{"id":"W4412513187","doi":"10.3390/diagnostics15141820","title":"Evaluating ChatGPT-4 Plus in Ophthalmology: Effect of Image Recognition and Domain-Specific Pretraining on Diagnostic Performance","year":2025,"lang":"en","type":"article","venue":"Diagnostics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Sherbrooke","funders":"Université de Sherbrooke","keywords":"Artificial intelligence; Ophthalmology; Domain (mathematical analysis); Computer science; Image (mathematics); Psychology; Optometry; Pattern recognition (psychology); Medicine; Computer vision; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008250198,0.001126912,0.0007822571,0.0007201047,0.0003152282,0.0009739881,0.001584684,0.001012945,0.00385747],"category_scores_gemma":[0.03783189,0.0003759882,0.0008828587,0.0003261869,0.0003783875,0.001156705,0.001970668,0.001104959,0.001505166],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001261746,"about_ca_system_score_gemma":0.001162833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003255246,"about_ca_topic_score_gemma":0.004065054,"domain_scores_codex":[0.9958718,0.002310052,0.0002927236,0.0007351727,0.0005784386,0.0002117311],"domain_scores_gemma":[0.9460897,0.04124055,0.003397457,0.002874469,0.003608934,0.002788982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01910833,0.01221298,0.4504061,0.001072177,0.000779721,0.0004404327,0.002621866,0.03764743,0.01359878,0.0002812472,0.008363348,0.4534675],"study_design_scores_gemma":[0.00150037,0.03563103,0.6175241,0.000391938,0.0009460862,0.0009805551,0.001459964,0.2945268,0.03821504,0.001059957,0.007439135,0.0003249342],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9920675,0.0001106738,0.004518907,0.000130838,0.00005172029,0.000557636,0.0003763626,0.0009094324,0.001277045],"genre_scores_gemma":[0.9820863,0.00009740923,0.01382221,0.0001623559,0.00004171108,0.0007109827,0.00127024,0.00005607575,0.001752673],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008250198,"threshold_uncertainty_score":0.04363173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1817417859717197,"score_gpt":0.456601610044809,"score_spread":0.2748598240730893,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}