{"id":"W4416792504","doi":"10.1038/s41433-025-04138-w","title":"Can ‘Deep Research’ agents and general AI agentic systems autonomously perform systematic review and meta-analysis?","year":2025,"lang":"en","type":"article","venue":"Eye","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"MEDLINE; Precision medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4443489,0.00235944,0.008606311,0.01235146,0.002132227,0.01245691,0.005128213,0.007035497,0.005100007],"category_scores_gemma":[0.68832,0.00314199,0.007987855,0.007420701,0.009190241,0.01613115,0.01165417,0.006707639,0.0008955832],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006600662,"about_ca_system_score_gemma":0.02938554,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004876197,"about_ca_topic_score_gemma":0.009082471,"domain_scores_codex":[0.5202317,0.4169844,0.03794542,0.01086719,0.01220002,0.001771287],"domain_scores_gemma":[0.1532283,0.7701563,0.02860209,0.03483852,0.01097488,0.002199973],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001972433,0.0001766493,0.01377288,0.1613064,0.07204766,0.0002033464,0.01232415,0.00715123,0.0009683612,0.2741979,0.03189113,0.4239878],"study_design_scores_gemma":[0.002719626,0.0006663695,0.003722947,0.06674596,0.02838012,0.000146961,0.002420434,0.01808521,0.0009904117,0.8119143,0.06391094,0.0002968035],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01689997,0.2689973,0.4309629,0.2477403,0.01070411,0.007134852,0.001506671,0.001756341,0.01429755],"genre_scores_gemma":[0.2882814,0.03492856,0.6188568,0.04023987,0.003007053,0.01247055,0.0006377975,0.000318064,0.001259989],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5556511,"threshold_uncertainty_score":0.6852168,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1005571165387217,"score_gpt":0.3872632382370159,"score_spread":0.2867061216982942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}