{"id":"W4410984428","doi":"10.22541/au.174897559.99564896/v1","title":"A Comparative Analysis of Artificial Intelligence Search Tools for Evidence Synthesis","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Artificial Intelligence in Healthcare","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Agency for Drugs and Technologies in Health","funders":"","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1555887,0.002797248,0.006182847,0.03452684,0.001202019,0.01130098,0.002606431,0.00327318,0.02184149],"category_scores_gemma":[0.5776654,0.00120389,0.009081642,0.02301861,0.002761676,0.009183271,0.005913442,0.002251864,0.001330087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006711536,"about_ca_system_score_gemma":0.007731282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003005592,"about_ca_topic_score_gemma":0.003750028,"domain_scores_codex":[0.6919808,0.2492102,0.02582605,0.004424979,0.02730863,0.001249314],"domain_scores_gemma":[0.1136161,0.8594415,0.00828733,0.008256083,0.009697222,0.0007018314],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.02129419,0.0008724103,0.01700547,0.05999257,0.02572242,0.0003809174,0.00259578,0.02176417,0.00118493,0.05577129,0.005564095,0.7878516],"study_design_scores_gemma":[0.01895273,0.02205474,0.0784246,0.07174467,0.1240923,0.004997509,0.006457707,0.2921957,0.009510176,0.2953796,0.07459795,0.001592233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.1069166,0.5303801,0.2984611,0.007436024,0.001039802,0.007450447,0.007094826,0.002028361,0.03919265],"genre_scores_gemma":[0.612086,0.05882496,0.3160447,0.001103729,0.0006372889,0.00484776,0.003565918,0.0005906777,0.002298963],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8444113,"threshold_uncertainty_score":0.8228415,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6558632414450133,"score_gpt":0.6182259202497302,"score_spread":0.03763732119528307,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}