{"id":"W4411259709","doi":"10.1101/2025.06.13.25329541","title":"Automation of Systematic Reviews with Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Ontario; University Health Network; Ottawa Hospital; University of Alberta; St. Michael's Hospital; University of British Columbia; Mount Sinai Hospital; University of Calgary; McGill University; University of Ottawa; Vector Institute; Wilfrid Laurier University; University of Waterloo; University of Toronto","funders":"","keywords":"Automation; Computer science; Systems engineering; Engineering; Mechanical engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3451886,0.004184995,0.005736387,0.01535628,0.003105143,0.01292777,0.005899989,0.002260901,0.008552042],"category_scores_gemma":[0.6902722,0.004769209,0.01028669,0.0108335,0.002289626,0.009202734,0.01472854,0.004320409,0.006057298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005466822,"about_ca_system_score_gemma":0.03691785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0063529,"about_ca_topic_score_gemma":0.01244848,"domain_scores_codex":[0.5759131,0.3150223,0.0631388,0.01722218,0.02706813,0.001635514],"domain_scores_gemma":[0.1652902,0.6655812,0.03305399,0.09049965,0.04416492,0.001410128],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002499025,0.0004562231,0.007842939,0.02769705,0.003904619,0.0006317784,0.008930924,0.03090745,0.0119561,0.01731339,0.03282169,0.8550388],"study_design_scores_gemma":[0.008225611,0.002113437,0.01752048,0.02438432,0.00847602,0.00126193,0.00489907,0.4366488,0.05243482,0.2164509,0.2254257,0.002158968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01235187,0.003160314,0.9286479,0.003226568,0.0004690925,0.01270865,0.00494645,0.03092823,0.003560966],"genre_scores_gemma":[0.03429917,0.0007932429,0.9499087,0.0005724138,0.0001252335,0.01012074,0.002459298,0.001298368,0.0004227515],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6548114,"threshold_uncertainty_score":0.8074991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03430894453732251,"score_gpt":0.3234808861615266,"score_spread":0.2891719416242041,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}