{"id":"W4409224572","doi":"10.1053/j.gastro.2025.03.034","title":"Large Language Model-Supported Systematic Reviews to Augment Clinical Guideline Development: An American Gastroenterological Association Pilot","year":2025,"lang":"en","type":"article","venue":"Gastroenterology","topic":"Clinical practice guidelines implementation","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"London Health Sciences Centre; Western University","funders":"National Institute of Diabetes and Digestive and Kidney Diseases; National Institutes of Health","keywords":"Augment; Guideline; Computer science; Medicine; Medical physics; Linguistics; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2123325,0.001318899,0.002382926,0.003634679,0.001977915,0.00543343,0.003492972,0.002759031,0.01063895],"category_scores_gemma":[0.5063009,0.001923039,0.005021357,0.004170937,0.0012648,0.007586606,0.008342833,0.004627486,0.001862984],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004452967,"about_ca_system_score_gemma":0.03768102,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007818152,"about_ca_topic_score_gemma":0.02350511,"domain_scores_codex":[0.8105403,0.1640565,0.01278548,0.003308186,0.008279339,0.001030205],"domain_scores_gemma":[0.2922952,0.5798022,0.02183114,0.05465036,0.04291368,0.008507414],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.05141662,0.01795192,0.02159927,0.05673467,0.01342428,0.0008216259,0.01801632,0.009221821,0.006918982,0.01034477,0.08007181,0.713478],"study_design_scores_gemma":[0.2629738,0.04764908,0.08381191,0.06743357,0.06303672,0.001644658,0.007397662,0.148274,0.01423549,0.05758503,0.2433188,0.002639352],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4421442,0.02662659,0.2422894,0.1177421,0.004663913,0.09854712,0.02352687,0.01687546,0.02758435],"genre_scores_gemma":[0.3303513,0.003719877,0.6068079,0.008850782,0.0004440022,0.04322343,0.004525931,0.0006721367,0.001404599],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7876675,"threshold_uncertainty_score":0.9713343,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1619749917263416,"score_gpt":0.4963791826588079,"score_spread":0.3344041909324663,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}