{"id":"W4409794823","doi":"10.2196/64963","title":"Comparing Diagnostic Accuracy of Clinical Professionals and Large Language Models: Systematic Review and Meta-Analysis","year":2025,"lang":"en","type":"review","venue":"JMIR Medical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"CINAHL; Triage; MEDLINE; Meta-analysis; Medicine; Risk assessment; Computer science; Medical emergency; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004158426,0.0003206898,0.009870959,0.0003167138,0.00006575859,0.00001962214,0.0002158098,0.0005191984,0.0002259839],"category_scores_gemma":[0.01705375,0.0001833572,0.001182475,0.0007099824,0.0001265888,0.0001524087,0.0002108733,0.0008515518,0.0000150918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002880214,"about_ca_system_score_gemma":0.001010238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002726004,"about_ca_topic_score_gemma":0.00001643176,"domain_scores_codex":[0.9929524,0.0005227418,0.005349973,0.000194244,0.0007303728,0.0002502387],"domain_scores_gemma":[0.9863944,0.01049698,0.001826995,0.0006028297,0.0002499289,0.0004288324],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.000001458878,0.0001323522,0.00001430453,0.9597361,0.01656532,0.000003917431,0.001108406,6.232425e-8,8.17877e-11,0.0001994545,0.0006964394,0.02154217],"study_design_scores_gemma":[0.00004372817,0.00004524217,0.000001785331,0.3218344,0.6696755,0.00003184643,0.001669048,0.001596621,5.040222e-8,0.00004109229,0.00489115,0.0001695829],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.00002575407,0.994377,0.0004640387,0.000426519,0.00007680675,0.00430421,0.00004733526,0.00002469101,0.0002536317],"genre_scores_gemma":[0.0002928149,0.9955075,0.0003191402,0.002525391,0.00005091451,0.0008097606,0.0002378592,0.00001180246,0.0002447504],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.6531101,"threshold_uncertainty_score":0.991226,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.457867284644542,"score_gpt":0.5990253850888093,"score_spread":0.1411581004442674,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}