{"id":"W4413040066","doi":"10.3233/shti250922","title":"Human in the Loop: Embedding Medical Expert Input in Large Language Models for Clinical Applications","year":2025,"lang":"en","type":"article","venue":"Studies in health technology and informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Automatic summarization; Ontology; Dravet syndrome; Epilepsy; Unified Medical Language System; Artificial intelligence; Natural language processing; Data science; Medicine; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002649942,0.00007724606,0.0002643148,0.0004184533,0.0001719936,0.00001093571,0.0006139586,0.0001766243,2.495012e-7],"category_scores_gemma":[0.000355377,0.00005850524,0.00001686821,0.0006436635,0.0001570994,0.0001593437,0.0004692929,0.0004575905,4.058367e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005760208,"about_ca_system_score_gemma":0.0001000487,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001509562,"about_ca_topic_score_gemma":0.0002912868,"domain_scores_codex":[0.9984973,0.00005216143,0.000933652,0.000130507,0.0001006891,0.000285663],"domain_scores_gemma":[0.9991568,0.0003401719,0.0001042495,0.0003547982,0.00002565638,0.00001836432],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001699509,0.00007408375,0.004860819,0.0003333586,0.000008153103,0.000001719528,0.02350228,0.0001704181,5.242241e-8,0.8760333,0.0004701534,0.09454399],"study_design_scores_gemma":[0.001115238,0.00006891542,0.0003982887,0.0003558373,0.000001101054,0.000004781989,0.03449355,0.8548414,0.000001432206,0.104072,0.004550239,0.00009730386],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05057677,0.008646575,0.9169606,0.02192116,0.0001736624,0.00103693,0.000001850147,0.00009974837,0.0005827414],"genre_scores_gemma":[0.937161,0.003740956,0.04958019,0.008731463,0.0000244774,0.0007397875,0.000001638593,0.000002906052,0.00001760321],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8865842,"threshold_uncertainty_score":0.2385774,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08934316528721445,"score_gpt":0.4816703637891037,"score_spread":0.3923271985018892,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}