{"id":"W7117138703","doi":"10.1186/s12916-025-04584-z","title":"Computer assisted verbal autopsy: comparing large language models to physicians for assigning causes to 6939 deaths in Sierra Leone from 2019–2022","year":2025,"lang":"en","type":"article","venue":"BMC Medicine","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Global Health Research; St. Michael's Hospital","funders":"Canadian Institutes of Health Research; Bill and Melinda Gates Foundation","keywords":"Sierra leone; Coding (social sciences); Quality (philosophy); MEDLINE; Quality management","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007876384,0.0002726788,0.0006591622,0.0004356404,0.0001715985,0.00006625734,0.0009908371,0.00009611603,0.0000173847],"category_scores_gemma":[0.0002713694,0.0002583207,0.0000665652,0.001017193,0.00002421354,0.0001926323,0.0005457966,0.0003716069,0.0000194627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002300452,"about_ca_system_score_gemma":0.0001731194,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003636756,"about_ca_topic_score_gemma":0.003763555,"domain_scores_codex":[0.9973297,0.0002348048,0.0005453543,0.0008235594,0.0004292536,0.0006373296],"domain_scores_gemma":[0.9978693,0.0008625152,0.0001191338,0.0008336871,0.0001121915,0.0002031468],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005150515,0.0008844983,0.1941419,0.001300668,0.0002786604,0.0002712924,0.1387017,0.2724151,0.01121112,0.06118264,0.1200649,0.1990326],"study_design_scores_gemma":[0.001813722,0.0002296545,0.09949861,0.00116188,0.00002028938,0.000002215297,0.0005254227,0.8938265,0.0001345537,0.0005133842,0.001993482,0.0002803236],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1281557,0.0003536229,0.8630304,0.005672039,0.001136416,0.000734232,0.00002125545,0.0002711091,0.0006251875],"genre_scores_gemma":[0.84804,0.000002184112,0.1428486,0.007862921,0.0005081199,0.0001206273,0.00006436664,0.0000254795,0.0005276621],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7201818,"threshold_uncertainty_score":0.9999869,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04734636144092669,"score_gpt":0.3481439785694647,"score_spread":0.300797617128538,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}