{"id":"W7130731173","doi":"10.1109/swc65939.2025.00239","title":"Small Language Models for Emergency Departments Decision Support: A Benchmark Study","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Rockyview General Hospital; University of Calgary","funders":"","keywords":"Benchmark (surveying); Workflow; Variety (cybernetics); Key (lock); Language model; Focus (optics); Clinical decision making","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001783261,0.0005499106,0.0006351898,0.0004913619,0.0007060613,0.0002652028,0.00232681,0.000221368,0.001262465],"category_scores_gemma":[0.0005055905,0.0005233398,0.0003121896,0.001292657,0.00002932567,0.0004826868,0.001506017,0.000471831,0.00008871927],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001923174,"about_ca_system_score_gemma":0.0006202577,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001087254,"about_ca_topic_score_gemma":0.002899686,"domain_scores_codex":[0.9945755,0.0004352602,0.0014872,0.001732736,0.0006691641,0.001100111],"domain_scores_gemma":[0.9962569,0.0005711763,0.0003107414,0.002051781,0.0004691591,0.0003402068],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003102548,0.003004618,0.1052792,0.0007422608,0.0003380172,0.0001097926,0.02057106,0.01216422,0.00001404704,0.03747499,0.01952566,0.8004659],"study_design_scores_gemma":[0.002005445,0.001828702,0.0130142,0.00008127615,0.00009760625,0.000004115255,0.001168457,0.9629157,0.00002885415,0.01594168,0.002387373,0.0005266545],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1256629,0.0009151699,0.8511128,0.001243429,0.004511801,0.004303923,0.00003014772,0.000209695,0.01201011],"genre_scores_gemma":[0.8894629,0.0000770183,0.08905886,0.0006774096,0.0001278617,0.0004149277,0.00002489748,0.00003347601,0.02012261],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9507514,"threshold_uncertainty_score":0.9997218,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04067461427722333,"score_gpt":0.3637823539918068,"score_spread":0.3231077397145834,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}