{"id":"W4392201401","doi":"10.2196/57594","title":"Correction: How Does ChatGPT Perform on the United States Medical Licensing Examination (USMLE)? The Implications of Large Language Models for Medical Education and Knowledge Assessment","year":2024,"lang":"en","type":"erratum","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Heart, Lung, and Blood Institute","keywords":"United States Medical Licensing Examination; Medical education; Licensure; Psychology; Medicine; Medical school","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005544985,0.002003862,0.001775268,0.003113177,0.003938592,0.003168745,0.003243625,0.01042541,0.04841116],"category_scores_gemma":[0.1293709,0.001181053,0.00157013,0.001991849,0.003053512,0.002226136,0.002151281,0.01585533,0.02404174],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004314387,"about_ca_system_score_gemma":0.00745579,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04380335,"about_ca_topic_score_gemma":0.04505096,"domain_scores_codex":[0.9941815,0.001120003,0.00133603,0.0006650989,0.002278797,0.0004185629],"domain_scores_gemma":[0.9536332,0.01928771,0.002498465,0.002094857,0.02113518,0.001350563],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00002197903,0.000004213311,0.00008454176,0.00006580591,0.000007818599,0.0004261041,0.00006567976,0.00002922532,0.00001891664,0.0005859623,0.9950688,0.003620898],"study_design_scores_gemma":[0.000085657,0.00002905134,0.001062282,0.0008005287,0.00005594241,0.002173581,0.0002793932,0.0005133924,0.0003293413,0.002729088,0.9918636,0.00007813892],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"editorial","genre_gemma":"empirical","genre_scores_codex":[0.0002936079,0.0007251991,0.0009701006,0.1923127,0.7991107,0.00003754246,0.00220045,0.0005875796,0.003762109],"genre_scores_gemma":[0.02570157,0.007901139,0.007309108,0.3272304,0.4396057,0.0003672761,0.004218874,0.002118456,0.1855475],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.994455,"threshold_uncertainty_score":0.1619515,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05352535385160018,"score_gpt":0.4603469147732917,"score_spread":0.4068215609216915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}