{"id":"W4323030608","doi":"10.1016/j.jfop.2023.100005","title":"Conversational AI Models for ophthalmic diagnosis: Comparison of ChatGPT and the Isabel Pro Differential Diagnosis Generator","year":2023,"lang":"en","type":"article","venue":"JFO Open Ophthalmology","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":116,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Toronto","funders":"","keywords":"Medical diagnosis; Differential diagnosis; Generator (circuit theory); Medicine; Set (abstract data type); Differential (mechanical device); Computer science; Artificial intelligence; Medical physics; Radiology; Pathology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0006681544,0.0002267233,0.0009992836,0.0001036,0.0001673627,0.00005645,0.0003616839,0.000236361,0.0004528195],"category_scores_gemma":[0.008476087,0.0001561588,0.0001908139,0.0002403789,0.0005364229,0.0001028265,0.0003952665,0.0002312725,0.00003149132],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000307958,"about_ca_system_score_gemma":0.0001754088,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003223964,"about_ca_topic_score_gemma":0.000003885962,"domain_scores_codex":[0.9980003,0.0001801637,0.00067414,0.00049962,0.0002700293,0.0003757616],"domain_scores_gemma":[0.9833765,0.01557804,0.0002394881,0.0004074383,0.0002184124,0.0001801508],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003664485,0.001993706,0.9183761,0.0002434753,0.0007068925,0.0001379578,0.001661257,0.0004345778,0.0001144695,0.02371624,0.04411707,0.004833778],"study_design_scores_gemma":[0.06288207,0.005796141,0.6649717,0.002184975,0.002009005,0.001545325,0.001502461,0.1927087,0.006613079,0.05668116,0.001951727,0.001153584],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9831918,0.0005739789,0.0001305858,0.01221493,0.0005500677,0.002501681,0.000153946,0.00003491473,0.0006481015],"genre_scores_gemma":[0.9941472,0.0002159083,0.0004010927,0.0005950711,0.0002381119,0.003462532,0.000244885,0.00003698365,0.0006581859],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2534044,"threshold_uncertainty_score":0.999876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1180761907008215,"score_gpt":0.4254146746001916,"score_spread":0.3073384838993701,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}