{"id":"W4393069045","doi":"10.2196/55048","title":"Exploring the Performance of ChatGPT Versions 3.5, 4, and 4 With Vision in the Chilean Medical Licensing Examination: Observational Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Observational study; Software deployment; Adaptability; Field (mathematics); Medical education; Psychology; Computer science; Medicine; Software engineering; Pathology; Management","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001727833,0.00007973013,0.0001101361,0.0001268343,0.0001462966,0.00002688741,0.0001102377,0.00005906949,0.0001527803],"category_scores_gemma":[0.0006163264,0.00004202491,0.0000178985,0.0005956006,0.0001507291,0.0002436673,0.00002139806,0.0004135019,0.000008408319],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006298952,"about_ca_system_score_gemma":0.001475817,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002300479,"about_ca_topic_score_gemma":0.0001427645,"domain_scores_codex":[0.9982086,0.0001474669,0.0003284982,0.000186102,0.001005743,0.0001235701],"domain_scores_gemma":[0.9989777,0.0005531693,0.00004346301,0.0001824989,0.0001261862,0.0001169878],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00006543025,0.001252403,0.1369778,0.0003246472,0.00001814082,0.000008917274,0.08051221,0.000003358768,0.00001750438,0.00114254,0.0009589061,0.7787181],"study_design_scores_gemma":[0.00007982718,0.0005982383,0.9488189,0.001469743,0.00002747198,0.00008792936,0.04104514,0.005927245,0.00006335219,0.00003622187,0.001788864,0.00005711513],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9569181,0.000273412,0.00001684822,0.04133094,0.0005499613,0.0006520918,2.468566e-7,0.0000177202,0.0002407104],"genre_scores_gemma":[0.9978026,0.0002465622,0.0000350638,0.001022819,0.0005865239,0.0002254681,0.00002092073,0.0000081411,0.00005191853],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.811841,"threshold_uncertainty_score":0.2618036,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2598105918815568,"score_gpt":0.4626373634131567,"score_spread":0.2028267715315999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}