{"id":"W4317910576","doi":"10.1101/2023.01.22.23284882","title":"Evaluating the Performance of ChatGPT in Ophthalmology: An Analysis of its Successes and Shortcomings","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":104,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre intégré universitaire de santé et de services sociaux de l'Est-de-l'Île-de-Montréal; Centre Hospitalier de l’Université de Montréal; Université de Montréal; Centre Intégré Universitaire de Santé et de Services Sociaux du Centre-Sud-de-l'Île-de-Montréal; Hôpital Maisonneuve-Rosemont","funders":"","keywords":"Recall; Interpretation (philosophy); Ophthalmic pathology; Space (punctuation); Medicine; Computer science; Optometry; Ophthalmology; Medical physics; Psychology; Neuro-ophthalmology; Glaucoma; Cognitive psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gpt","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"grok","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"opus","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0249403,0.001495576,0.00122723,0.002119887,0.0005106012,0.001904264,0.002106939,0.001954654,0.002014229],"category_scores_gemma":[0.1082624,0.0004147825,0.0007823037,0.001100039,0.0008219641,0.002824464,0.003627396,0.001595521,0.001986861],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001021006,"about_ca_system_score_gemma":0.001277847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005451503,"about_ca_topic_score_gemma":0.003613631,"domain_scores_codex":[0.978584,0.01576101,0.001178405,0.002099859,0.001989326,0.000387372],"domain_scores_gemma":[0.8067339,0.1702754,0.003672566,0.00585622,0.01059275,0.002869162],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01040694,0.001742855,0.2352705,0.002756161,0.001121357,0.0007302486,0.006714917,0.07277751,0.02401816,0.001047,0.01296361,0.6304507],"study_design_scores_gemma":[0.0005757274,0.004603134,0.1043125,0.0003943045,0.0006032243,0.001164554,0.002573912,0.841878,0.03282829,0.003200175,0.007633055,0.0002331517],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9200632,0.001681181,0.06091364,0.001290366,0.000175842,0.0006308674,0.002652664,0.008832011,0.003760313],"genre_scores_gemma":[0.9615008,0.0002122826,0.03335012,0.0003063138,0.00006874171,0.0003537,0.002794601,0.0002351386,0.001178278],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9750597,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3361547074696869,"score_gpt":0.5181460555126232,"score_spread":0.1819913480429363,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}