{"id":"W4387653139","doi":"10.1080/0142159x.2023.2249588","title":"ChatGPT-4: An assessment of an upgraded artificial intelligence chatbot in the United States Medical Licensing Examination","year":2023,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":112,"is_retracted":false,"has_abstract":true,"ca_institutions":"St. Michael's Hospital; University of Toronto","funders":"","keywords":"United States Medical Licensing Examination; Licensure; Medicine; Medical education; Multiple choice; Test (biology); Family medicine; Psychology; Medical school; Significant difference; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006859775,0.0001458356,0.0002790307,0.0004162654,0.0001030907,0.00002475925,0.0002982549,0.0003193873,0.001427833],"category_scores_gemma":[0.002571015,0.0001034424,0.00004816072,0.001429869,0.0002900197,0.0001350157,0.00003728599,0.0007923517,0.00004044091],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001047023,"about_ca_system_score_gemma":0.0005175897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002428568,"about_ca_topic_score_gemma":0.0009090867,"domain_scores_codex":[0.9959104,0.0008460074,0.0008187011,0.000323309,0.001714917,0.0003866518],"domain_scores_gemma":[0.9985309,0.0003942561,0.0001195525,0.0003854781,0.0001680514,0.0004018125],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007271669,0.002149372,0.02549204,0.0001508877,0.00002545896,0.0001939853,0.04706738,0.00008560895,0.000129144,0.00492698,0.0004050551,0.9193014],"study_design_scores_gemma":[0.00007168193,0.0007575847,0.0643416,0.0003308643,0.00003028923,0.00004455633,0.04840131,0.8794417,0.001097978,0.004359464,0.0009562679,0.0001667018],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9788063,0.00002342464,0.002060666,0.01794873,0.0003495892,0.000385234,0.000001011323,0.00009302259,0.0003320952],"genre_scores_gemma":[0.9971443,0.0001703819,0.0001465397,0.001265377,0.0006821754,0.00004352621,0.0004561232,0.00002077342,0.00007081469],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9191347,"threshold_uncertainty_score":0.999485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2621084780040356,"score_gpt":0.5165753969505843,"score_spread":0.2544669189465487,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}