{"id":"W4381687070","doi":"10.2196/48305","title":"Variability in Large Language Models’ Responses to Medical Licensing and Certification Examinations. Comment on “How Does ChatGPT Perform on the United States Medical Licensing Examination? The Implications of Large Language Models for Medical Education and Knowledge Assessment”","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Certification; United States Medical Licensing Examination; Medical education; Medical school; Medicine; Political science; Law","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01100927,0.0002574036,0.0003572602,0.000664464,0.0005317423,0.00007176438,0.0003261203,0.0004647049,0.0003720288],"category_scores_gemma":[0.02275671,0.0001642337,0.00006208432,0.0012868,0.0002916264,0.0002050195,0.0001294886,0.0007795658,0.000007898748],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004100324,"about_ca_system_score_gemma":0.005267234,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007147851,"about_ca_topic_score_gemma":0.001103955,"domain_scores_codex":[0.9953567,0.00109667,0.0008965306,0.0005969009,0.001572824,0.0004803307],"domain_scores_gemma":[0.9911512,0.006401675,0.0002405047,0.0006662881,0.0006947148,0.0008456027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003295482,0.007880133,0.01029245,0.0008818936,0.00007318969,0.000003356605,0.1540163,0.00002109318,0.0001227352,0.1135398,0.02206417,0.6907753],"study_design_scores_gemma":[0.0007087596,0.0004689245,0.1407878,0.002611727,0.00009061709,0.00004009541,0.1300328,0.7107203,0.0005171623,0.00704263,0.006597487,0.0003816932],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6633213,0.00009887541,0.002155078,0.332041,0.0003809716,0.001676474,0.00002551908,0.00006045147,0.0002403581],"genre_scores_gemma":[0.9773172,0.0005059227,0.0001752756,0.01885535,0.0005108544,0.001392544,0.001005955,0.00003628302,0.0002005971],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7106992,"threshold_uncertainty_score":0.985475,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08613425416401634,"score_gpt":0.469902279629843,"score_spread":0.3837680254658267,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}