{"id":"W4415396422","doi":"10.1002/cesm.70042","title":"Enhancing Evidence Synthesis Efficiency: Leveraging Large Language Models and Agentic Workflows for Optimized Literature Screening","year":2025,"lang":"en","type":"article","venue":"Cochrane Evidence Synthesis and Methods","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Agency of Canada; University of Waterloo; Health Canada","funders":"Public Health Agency; World Health Organization","keywords":"Workflow; Public health; Key (lock); Language model; Health care; Component (thermodynamics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04331007,0.00295871,0.002581374,0.008356158,0.001433438,0.007730145,0.003657051,0.002437432,0.005290871],"category_scores_gemma":[0.1952901,0.001942994,0.004155758,0.004242939,0.001330714,0.006564461,0.006208037,0.003971759,0.0025672],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002846641,"about_ca_system_score_gemma":0.01420338,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01185344,"about_ca_topic_score_gemma":0.02417278,"domain_scores_codex":[0.9729131,0.01912871,0.003114968,0.002293976,0.002205908,0.0003431709],"domain_scores_gemma":[0.7915399,0.1807104,0.007073562,0.01027417,0.008270091,0.002131899],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002184134,0.0007317484,0.0142306,0.007598013,0.00224334,0.0008693864,0.004202525,0.1728343,0.01334498,0.01036648,0.02253408,0.7488604],"study_design_scores_gemma":[0.000736256,0.0003000529,0.001701581,0.0008151531,0.0007981632,0.0003025057,0.0006250127,0.9023995,0.01005462,0.06186635,0.02020749,0.0001932179],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02108413,0.003766893,0.9323997,0.00561228,0.0002602991,0.001851058,0.00252588,0.02989986,0.002599872],"genre_scores_gemma":[0.08121847,0.0006610501,0.9135609,0.0008964271,0.0001035719,0.0009554005,0.001545298,0.0005361806,0.000522801],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.95669,"threshold_uncertainty_score":0.2290483,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1271447016142044,"score_gpt":0.4821077098525858,"score_spread":0.3549630082383814,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}