{"id":"W4386788271","doi":"10.1101/2023.09.14.23295571","title":"Comparing the Performance of ChatGPT and GPT-4 versus a Cohort of Medical Students on an Official University of Toronto Undergraduate Medical Education Progress Test","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph; University of Toronto","funders":"","keywords":"McNemar's test; Cohort; Test (biology); Medical education; Medicine; Medical school; Cohort study; Psychology; Mathematics education; Internal medicine; Statistics; Mathematics; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003462742,0.0008176044,0.0006367532,0.001331537,0.0005720775,0.0009374285,0.0007510257,0.0007051619,0.003871544],"category_scores_gemma":[0.01510278,0.0003023726,0.000674019,0.0005988785,0.0006673434,0.0007246037,0.001795456,0.0007482073,0.001797707],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001687378,"about_ca_system_score_gemma":0.001260839,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04385137,"about_ca_topic_score_gemma":0.07259024,"domain_scores_codex":[0.9977583,0.000409295,0.0002630583,0.0004213543,0.0007200216,0.0004279044],"domain_scores_gemma":[0.9830064,0.00356117,0.00284072,0.000968139,0.004070104,0.005553398],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002909177,0.001298559,0.9500201,0.0001175193,0.0001316898,0.0001790863,0.004270288,0.0007692461,0.002087266,0.00006936774,0.004662982,0.03348473],"study_design_scores_gemma":[0.00005027273,0.001932624,0.9937092,0.00001673601,0.00003081457,0.00007051439,0.001114409,0.001220228,0.0009368106,0.00002010611,0.0008768937,0.00002141115],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9976329,0.00004945242,0.00009693494,0.00006982735,0.00002722044,0.0000764315,0.0005682865,0.00003944903,0.001439468],"genre_scores_gemma":[0.9948459,0.00005535297,0.0003362786,0.00005115501,0.00001988291,0.00008772786,0.001505443,0.00001435504,0.003083836],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9983126,"threshold_uncertainty_score":0.08719224,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1164460701808062,"score_gpt":0.4285416007085918,"score_spread":0.3120955305277856,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}