{"id":"W2901757934","doi":"10.1503/cjs.015917","title":"Effect of rater training on the reliability of technical skill assessments: a randomized controlled trial","year":2018,"lang":"en","type":"article","venue":"Canadian Journal of Surgery","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Medicine; Checklist; Inter-rater reliability; Intraclass correlation; Reliability (semiconductor); Randomized controlled trial; Physical therapy; Intra-rater reliability; Visual analogue scale; Confidence interval; Observational study; Rating scale; Medical physics; Physical medicine and rehabilitation; Surgery; Psychometrics; Statistics; Psychology; Clinical psychology; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01252792,0.0001150543,0.001903324,0.0002797784,0.00005241541,0.00001027113,0.00007338452,0.0001208586,0.0008780863],"category_scores_gemma":[0.01604271,0.00005575221,0.0009191968,0.0001927724,0.0004481639,0.00004452928,0.000003497324,0.0003319755,0.000001556431],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007339991,"about_ca_system_score_gemma":0.00110839,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001158054,"about_ca_topic_score_gemma":0.00005951995,"domain_scores_codex":[0.997187,0.001044019,0.001171409,0.00008999139,0.0003126581,0.0001949547],"domain_scores_gemma":[0.9819891,0.01649287,0.0006512899,0.0001688401,0.0003763615,0.0003215614],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"randomized_trial","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.9924134,0.00004988442,0.002580995,0.0000230072,0.0002711751,0.00006059583,0.0001915101,0.00001396193,0.00009350853,0.0001667511,0.0002972803,0.003837924],"study_design_scores_gemma":[0.9928529,0.001331864,0.002527024,0.0005485257,0.000378323,0.00003852561,0.00008585566,0.0001929938,0.0007662646,0.00009935884,0.001104217,0.00007420148],"study_design_candidate":"randomized_trial","study_design_consensus":"randomized_trial","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9914341,0.00004313814,0.00004950017,0.0009754807,0.0006598637,0.0008795268,0.000002653557,0.000003692787,0.005952045],"genre_scores_gemma":[0.9992932,0.000004812031,0.00002677582,0.0003021077,0.0003040982,0.00001348132,0.000001174399,0.000009891492,0.00004443578],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01544885,"threshold_uncertainty_score":0.9922456,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04436002821238123,"score_gpt":0.3359270642908207,"score_spread":0.2915670360784395,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}