{"id":"W2592866053","doi":"10.1186/s13643-017-0441-7","title":"Effect of standardized training on the reliability of the Cochrane risk of bias assessment tool: a prospective study","year":2017,"lang":"en","type":"article","venue":"Systematic Reviews","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":65,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; St. Michael's Hospital; University of Toronto","funders":"","keywords":"Medicine; Kappa; Reliability (semiconductor); Physical therapy; Randomized controlled trial; Cohen's kappa; Risk assessment; MEDLINE; Clinical psychology; Statistics; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.433074,0.001755508,0.00523106,0.005035989,0.001305317,0.003316375,0.002928259,0.002939075,0.003024753],"category_scores_gemma":[0.7264251,0.002652823,0.009298286,0.006887711,0.004945758,0.005723073,0.003862865,0.00235193,0.0005882411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004363573,"about_ca_system_score_gemma":0.008418731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001557113,"about_ca_topic_score_gemma":0.001731764,"domain_scores_codex":[0.3386748,0.4839733,0.1169342,0.01670979,0.04122595,0.002481927],"domain_scores_gemma":[0.1176713,0.6045773,0.1891591,0.04751734,0.03888532,0.002189668],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.06563739,0.004280526,0.5518354,0.07209261,0.03524795,0.0003687745,0.01005033,0.004105384,0.001541391,0.002330682,0.00366244,0.2488472],"study_design_scores_gemma":[0.02563669,0.06710579,0.7994869,0.03549402,0.03722257,0.001723485,0.002551067,0.01241925,0.003222152,0.003741033,0.01051756,0.0008794632],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7669736,0.1112863,0.05233425,0.002178684,0.0009206143,0.05802941,0.002247431,0.0004128271,0.005616973],"genre_scores_gemma":[0.936367,0.00455802,0.02764271,0.0005727383,0.000235505,0.02975115,0.0006008619,0.00007608551,0.0001958466],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.566926,"threshold_uncertainty_score":0.6991208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2307356376426931,"score_gpt":0.4609718262253066,"score_spread":0.2302361885826134,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}