{"id":"W4392201401","doi":"10.2196/57594","title":"Correction: How Does ChatGPT Perform on the United States Medical Licensing Examination (USMLE)? The Implications of Large Language Models for Medical Education and Knowledge Assessment","year":2024,"lang":"en","type":"erratum","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Heart, Lung, and Blood Institute","keywords":"United States Medical Licensing Examination; Medical education; Licensure; Psychology; Medicine; Medical school","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["research_integrity","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00363874,0.0003870663,0.0004956303,0.0005363476,0.0005460976,0.0001236327,0.0004219899,0.0012159,0.001157929],"category_scores_gemma":[0.007339608,0.0002161839,0.0001704636,0.0008947365,0.0004349271,0.0001496792,0.0001057154,0.002365425,0.00001797008],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006113921,"about_ca_system_score_gemma":0.02465913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007026178,"about_ca_topic_score_gemma":0.001213533,"domain_scores_codex":[0.9957953,0.0004111993,0.000928698,0.000648643,0.001771985,0.0004441703],"domain_scores_gemma":[0.9947042,0.002113354,0.0004428188,0.0007248368,0.001330781,0.0006840261],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002444554,0.001353675,0.00008387872,0.0009661917,0.00006354524,5.352095e-7,0.01450063,4.837939e-7,0.000002000069,0.008246347,0.6814733,0.293285],"study_design_scores_gemma":[0.0002403074,0.0006288221,0.004585383,0.01117257,0.0005063986,0.0001758594,0.08678919,0.2586712,0.000118029,0.006136896,0.6304719,0.0005034667],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.09936476,0.006256553,0.002512927,0.7889683,0.08375455,0.006048829,0.00008830714,0.0002323697,0.01277339],"genre_scores_gemma":[0.8709764,0.008611706,0.000129466,0.01835627,0.01817643,0.005047433,0.01095801,0.0001656445,0.06757856],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7716117,"threshold_uncertainty_score":0.9999362,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05352535385160018,"score_gpt":0.4603469147732917,"score_spread":0.4068215609216915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}