{"id":"W4413584208","doi":"10.1016/j.jclinepi.2025.111944","title":"Use of artificial intelligence to support the assessment of the methodological quality of systematic reviews","year":2025,"lang":"en","type":"article","venue":"Journal of Clinical Epidemiology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Systematic review; Quality assessment; MEDLINE; Medicine; Psychology; Management science; Engineering; Political science; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3767141,0.004295995,0.01027327,0.03115606,0.002128761,0.01608912,0.004546168,0.003732121,0.002594221],"category_scores_gemma":[0.8094726,0.002703217,0.009368875,0.01545898,0.003909024,0.007325861,0.009529337,0.006346678,0.0003670492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004520033,"about_ca_system_score_gemma":0.01400052,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00682128,"about_ca_topic_score_gemma":0.008083698,"domain_scores_codex":[0.4733979,0.4207045,0.06731114,0.009631844,0.02820062,0.0007539491],"domain_scores_gemma":[0.04679479,0.9039466,0.02446201,0.01352875,0.01047145,0.0007963097],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004442638,0.0007967982,0.04293514,0.05901398,0.07404364,0.0006339637,0.004362589,0.1265574,0.002257631,0.0441951,0.01101509,0.6297461],"study_design_scores_gemma":[0.003000719,0.001274249,0.01081607,0.0144854,0.03135057,0.0006199787,0.0006285842,0.6558112,0.004182464,0.2620894,0.01500943,0.0007319999],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03372394,0.04348426,0.8841182,0.01802551,0.001345017,0.005725322,0.002435187,0.004742708,0.006399942],"genre_scores_gemma":[0.1474975,0.003843272,0.8434666,0.00121555,0.0003853017,0.002760126,0.0005199158,0.0001508023,0.0001608517],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6232859,"threshold_uncertainty_score":0.7686225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9941548078090001,"score_gpt":0.8082536389011452,"score_spread":0.1859011689078549,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}