{"id":"W4416792504","doi":"10.1038/s41433-025-04138-w","title":"Can ‘Deep Research’ agents and general AI agentic systems autonomously perform systematic review and meta-analysis?","year":2025,"lang":"en","type":"article","venue":"Eye","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"MEDLINE; Precision medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003080734,0.0002165443,0.001300351,0.0003996714,0.0004136218,0.0007265334,0.0008487047,0.00006420095,0.00005424809],"category_scores_gemma":[0.0002658741,0.0001657379,0.0002535266,0.001794754,0.0001182494,0.0003792188,0.0005107627,0.0002385327,0.00006168395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001243283,"about_ca_system_score_gemma":0.0000985494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007894923,"about_ca_topic_score_gemma":0.0002816264,"domain_scores_codex":[0.9969725,0.0006952666,0.0006926549,0.0006569461,0.000495825,0.0004867844],"domain_scores_gemma":[0.9981052,0.0001596761,0.0001426859,0.001113796,0.0003085155,0.0001701383],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006688911,0.0002678006,0.001046769,0.3437037,0.1001021,0.0003033197,0.005842627,0.001515666,0.0001495023,0.5373079,0.009095826,0.0006580007],"study_design_scores_gemma":[0.000164078,0.0001625489,0.0006144977,0.002664902,0.1897738,0.0000300998,0.0007132639,0.7970487,0.0005012753,0.006055102,0.001417199,0.0008545907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.0385265,0.7662815,0.1215954,0.0508247,0.00100903,0.01279442,0.00003355061,0.0006668237,0.008268068],"genre_scores_gemma":[0.9758137,0.001717411,0.0009903522,0.00304771,0.00002650657,0.0005209083,0.000004385601,0.00001477969,0.01786422],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9372872,"threshold_uncertainty_score":0.7005978,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1005571165387217,"score_gpt":0.3872632382370159,"score_spread":0.2867061216982942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}