{"id":"W4368340908","doi":"10.1016/j.xops.2023.100324","title":"Evaluating the Performance of ChatGPT in Ophthalmology","year":2023,"lang":"en","type":"article","venue":"Ophthalmology Science","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":494,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre Hospitalier de l’Université de Montréal; Université de Montréal; Centre Intégré Universitaire de Santé et de Services Sociaux du Centre-Sud-de-l'Île-de-Montréal; Hôpital Maisonneuve-Rosemont","funders":"American Academy of Ophthalmology; Bayer","keywords":"Logistic regression; Test (biology); Set (abstract data type); Computer science; Post hoc; Artificial intelligence; Index (typography); Regression; Scale (ratio); Section (typography); Statistics; Test set; Machine learning; Regression analysis; Key (lock); Natural language processing; Medicine; Mathematics; World Wide Web; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01408115,0.001729946,0.0008274328,0.001123677,0.0004621415,0.001815662,0.00196891,0.001865682,0.003039098],"category_scores_gemma":[0.08301011,0.0005032091,0.0007420124,0.0005472153,0.0006426625,0.002977915,0.003264416,0.001849795,0.001512121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001632736,"about_ca_system_score_gemma":0.00143215,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007774163,"about_ca_topic_score_gemma":0.007315191,"domain_scores_codex":[0.9896715,0.006850978,0.0006320106,0.001685187,0.0008800082,0.0002803247],"domain_scores_gemma":[0.897964,0.08856722,0.003444879,0.004019857,0.003915997,0.002088142],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01089531,0.003062688,0.2466386,0.001364184,0.0009732219,0.0007141256,0.004981062,0.1499874,0.01191545,0.001796378,0.01217717,0.5554944],"study_design_scores_gemma":[0.0004982813,0.003303141,0.04870565,0.0001545297,0.0002642357,0.0004280948,0.001004581,0.9284833,0.01015412,0.002986073,0.003861142,0.0001568481],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9466833,0.0005227988,0.04089529,0.0007257602,0.0001248262,0.0006077162,0.0009703255,0.005605813,0.003864265],"genre_scores_gemma":[0.9665953,0.0001112339,0.02922159,0.0002946264,0.00004248569,0.0003907644,0.001484711,0.0001718724,0.00168739],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01408115,"threshold_uncertainty_score":0.07446915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3596977566977329,"score_gpt":0.5413077638481226,"score_spread":0.1816100071503898,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}