{"id":"W4393069045","doi":"10.2196/55048","title":"Exploring the Performance of ChatGPT Versions 3.5, 4, and 4 With Vision in the Chilean Medical Licensing Examination: Observational Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Observational study; Software deployment; Adaptability; Field (mathematics); Medical education; Psychology; Computer science; Medicine; Software engineering; Pathology; Management","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006752389,0.0004074834,0.0006047163,0.001526299,0.000816367,0.001365812,0.001067805,0.0008642421,0.00157561],"category_scores_gemma":[0.04293839,0.000513633,0.0005476829,0.0009719659,0.001085473,0.001474212,0.002658383,0.001230705,0.0005737993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002528046,"about_ca_system_score_gemma":0.002661457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0175333,"about_ca_topic_score_gemma":0.02145946,"domain_scores_codex":[0.995566,0.001877425,0.0005905295,0.0004395081,0.0009113605,0.0006151448],"domain_scores_gemma":[0.9769154,0.007726443,0.007221214,0.001278623,0.004431225,0.002427206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003388687,0.001261764,0.9486355,0.0003154244,0.0000660742,0.0005270824,0.02884644,0.000233232,0.0008231294,0.00008682805,0.001044677,0.01782097],"study_design_scores_gemma":[0.00002673779,0.0009325375,0.9669767,0.0001281952,0.00003083347,0.0003044949,0.02824738,0.0008383246,0.0004234526,0.00005432747,0.001983135,0.00005376566],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9992381,0.00006452415,0.00005609795,0.00004990597,0.000003175617,0.00008962391,0.00008942047,0.000004951256,0.0004040004],"genre_scores_gemma":[0.9984449,0.0001178284,0.0003681782,0.00005193648,0.000007406466,0.0001958627,0.0002187064,0.000007103423,0.0005880664],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0175333,"threshold_uncertainty_score":0.03571045,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2598105918815568,"score_gpt":0.4626373634131567,"score_spread":0.2028267715315999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}