{"id":"W2901757934","doi":"10.1503/cjs.015917","title":"Effect of rater training on the reliability of technical skill assessments: a randomized controlled trial","year":2018,"lang":"en","type":"article","venue":"Canadian Journal of Surgery","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Medicine; Checklist; Inter-rater reliability; Intraclass correlation; Reliability (semiconductor); Randomized controlled trial; Physical therapy; Intra-rater reliability; Visual analogue scale; Confidence interval; Observational study; Rating scale; Medical physics; Physical medicine and rehabilitation; Surgery; Psychometrics; Statistics; Psychology; Clinical psychology; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01577251,0.002456009,0.007246822,0.001133021,0.001402092,0.00186433,0.002284097,0.005932846,0.0112614],"category_scores_gemma":[0.0329438,0.001599493,0.003356953,0.001102964,0.002946479,0.002533956,0.001222753,0.004296558,0.00145089],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0019743,"about_ca_system_score_gemma":0.00350884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00181383,"about_ca_topic_score_gemma":0.001917272,"domain_scores_codex":[0.9776154,0.01587363,0.001824761,0.002341775,0.001367063,0.0009773127],"domain_scores_gemma":[0.9812995,0.01042839,0.004051774,0.001308428,0.001129008,0.001782974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"randomized_trial","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.9758789,0.007528234,0.0003369555,0.001833369,0.000976896,0.00002156298,0.00009774137,0.0001441334,0.0003477987,0.00008879479,0.0004458702,0.01229978],"study_design_scores_gemma":[0.9613191,0.03648273,0.0006710578,0.0002112784,0.0005455667,0.000009782018,0.00002222519,0.0002493286,0.0001239896,0.0001009231,0.0002497949,0.00001422203],"study_design_candidate":"randomized_trial","study_design_consensus":"randomized_trial","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9106561,0.01980331,0.004257288,0.002359951,0.004822654,0.05222508,0.001037902,0.0006089676,0.004228748],"genre_scores_gemma":[0.9030549,0.005335834,0.009122686,0.001709171,0.002349753,0.075383,0.0003754777,0.00006372668,0.002605419],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01577251,"threshold_uncertainty_score":0.08341396,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04436002821238123,"score_gpt":0.3359270642908207,"score_spread":0.2915670360784395,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}