{"id":"W4416637138","doi":"10.1183/13993003.congress-2025.pa6248","title":"Comparative evaluation of ChatGPT-4, Claude 3.5 Sonnet, and Gemini 1.5 Advanced for patient education on chronic obstructive pulmonary disease (COPD): a global expert assessment of artificial intelligence (AI)-generated responses","year":2025,"lang":"","type":"article","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Hospital Foundation","funders":"","keywords":"Pulmonologists; Pulmonary disease; COPD; Likert scale; Test (biology); Bonferroni correction; Terminology; Quality of life (healthcare)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04686691,0.0005374144,0.0009660886,0.001530524,0.0004656933,0.001463045,0.0009286344,0.001033315,0.003267556],"category_scores_gemma":[0.1209396,0.0003124573,0.001101673,0.001115325,0.000939639,0.001638904,0.002330651,0.001136351,0.0007198176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002193241,"about_ca_system_score_gemma":0.002009076,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002258032,"about_ca_topic_score_gemma":0.002489293,"domain_scores_codex":[0.9595404,0.03058142,0.002943443,0.001510125,0.004637204,0.0007873788],"domain_scores_gemma":[0.8068928,0.153393,0.009796382,0.004335832,0.01782919,0.007752696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.08535836,0.01825002,0.4515521,0.006011683,0.001545171,0.0005248335,0.04552492,0.006692314,0.005414006,0.001573541,0.01213097,0.3654221],"study_design_scores_gemma":[0.004049363,0.06807932,0.883899,0.0006434994,0.000810888,0.0002625535,0.01183721,0.01351693,0.004279766,0.0006822946,0.01171942,0.0002196895],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9935455,0.0001435062,0.001131188,0.0002875302,0.00005884013,0.001320298,0.0005281175,0.0001033567,0.002881596],"genre_scores_gemma":[0.9861454,0.0002054378,0.006186982,0.0004035271,0.0000653173,0.004393726,0.001257709,0.00004227627,0.001299596],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04686691,"threshold_uncertainty_score":0.2478589,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1504163355014659,"score_gpt":0.5047888802377128,"score_spread":0.354372544736247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}