{"id":"W4414078785","doi":"10.1097/ijg.0000000000002627","title":"Comparing Performance of Large Language Model-Based Tools on Patient-Driven Glaucoma Inquiries","year":2025,"lang":"en","type":"article","venue":"Journal of Glaucoma","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Care Foundation","funders":"","keywords":"Glaucoma; Public health; MEDLINE; Occupational safety and health","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075876,0.00103924,0.0007029129,0.001510416,0.0002810819,0.001794882,0.000877758,0.001264878,0.002376215],"category_scores_gemma":[0.06683367,0.000328766,0.0009013188,0.0007359336,0.0003937751,0.002183507,0.001996998,0.001069088,0.001197034],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009493311,"about_ca_system_score_gemma":0.0008768911,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003022456,"about_ca_topic_score_gemma":0.002992476,"domain_scores_codex":[0.9940451,0.003692311,0.000671013,0.0007714714,0.0006431506,0.0001769822],"domain_scores_gemma":[0.9223605,0.06921506,0.002149984,0.002200097,0.002775561,0.001298784],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.02903271,0.008273866,0.2053917,0.004408644,0.001518276,0.001052825,0.01069501,0.0574594,0.01848694,0.00147872,0.01549792,0.6467039],"study_design_scores_gemma":[0.002149765,0.02049267,0.2262257,0.0007435505,0.001332686,0.001431515,0.006242807,0.6937025,0.02961391,0.005317957,0.01199856,0.0007483435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9845706,0.0002565207,0.007890118,0.000309129,0.00007645738,0.0004830458,0.001770197,0.002942834,0.001701157],"genre_scores_gemma":[0.9779573,0.0001557185,0.01705339,0.0001984535,0.00003745228,0.000499657,0.003040956,0.0001503989,0.0009066768],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0075876,"threshold_uncertainty_score":0.04012752,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08528478988284796,"score_gpt":0.3906838551475755,"score_spread":0.3053990652647276,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}