{"id":"W3006673017","doi":"10.1002/jrsm.1398","title":"Comparing machine and human reviewers to evaluate the risk of bias in randomized controlled trials","year":2020,"lang":"en","type":"article","venue":"Research Synthesis Methods","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Capital District Health Authority; Institute of Health Economics; University of Alberta","funders":"Canadian Institutes of Health Research; Institute of Health Economics; Alberta Innovates - Health Solutions; Physiotherapy Foundation of Canada; Government of Alberta","keywords":"Blinding; Randomized controlled trial; Computer science; Medical physics; MEDLINE; Sample size determination; Medicine; Applied psychology; Psychology; Statistics; Surgery; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7848458,0.005629208,0.01979201,0.02376086,0.003698847,0.01023811,0.0088757,0.008672431,0.00649429],"category_scores_gemma":[0.9142594,0.004355528,0.02175454,0.01456985,0.01183824,0.01290519,0.009210836,0.004767406,0.001757358],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009187241,"about_ca_system_score_gemma":0.02108887,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002513631,"about_ca_topic_score_gemma":0.003974174,"domain_scores_codex":[0.07284974,0.6871842,0.1735457,0.02232218,0.04281292,0.001285271],"domain_scores_gemma":[0.02214727,0.8183072,0.07713394,0.0262763,0.05485127,0.001284006],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.03193715,0.0006159931,0.06777691,0.3255582,0.116304,0.000675386,0.02188729,0.01123479,0.00281642,0.01347298,0.02305319,0.3846677],"study_design_scores_gemma":[0.06034574,0.01735148,0.1062053,0.248969,0.1488122,0.004379055,0.007365734,0.1473878,0.01709577,0.1283674,0.1083698,0.005350796],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07164908,0.1583907,0.5753759,0.01313721,0.01430207,0.1465588,0.004919387,0.004069203,0.01159779],"genre_scores_gemma":[0.3035186,0.009777252,0.5185442,0.00392135,0.002228083,0.158684,0.001500024,0.0008119725,0.001014616],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2151542,"threshold_uncertainty_score":0.2653235,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9609323859410629,"score_gpt":0.7227307998086823,"score_spread":0.2382015861323806,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}