{"id":"W2007544525","doi":"10.1097/sla.0b013e318160b371","title":"Toward Feasible, Valid, and Reliable Video-Based Assessments of Technical Surgical Skills in the Operating Room","year":2008,"lang":"en","type":"article","venue":"Annals of Surgery","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":177,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; St. Michael's Hospital","funders":"","keywords":"Medicine; Checklist; Rating scale; Reliability (semiconductor); Summative assessment; Medical physics; Scale (ratio); Physical therapy; Statistics; Psychology; Formative assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03849045,0.0005212929,0.0003591435,0.00114433,0.0003503307,0.001260617,0.0009653281,0.0009491394,0.0008804574],"category_scores_gemma":[0.1082913,0.000338586,0.0003265768,0.0004507251,0.0008778133,0.00174742,0.001491863,0.0008590512,0.0004730954],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006011417,"about_ca_system_score_gemma":0.002085292,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00098691,"about_ca_topic_score_gemma":0.001316958,"domain_scores_codex":[0.9773916,0.0124797,0.001452907,0.001214093,0.007016588,0.0004450864],"domain_scores_gemma":[0.9250413,0.03728184,0.01122962,0.002919253,0.02209073,0.001437298],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00293302,0.002811749,0.6559203,0.001282475,0.0001395301,0.0002396626,0.00516863,0.002385777,0.03723726,0.0005988315,0.001629376,0.2896534],"study_design_scores_gemma":[0.0008965106,0.01830597,0.9432107,0.0006263469,0.000118443,0.0005909856,0.003794621,0.01541194,0.01301089,0.0006466429,0.003292307,0.00009467378],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9661142,0.0005636306,0.02895603,0.0003431737,0.00008156385,0.001766754,0.0001461823,0.00007729346,0.001951232],"genre_scores_gemma":[0.9161718,0.0003271269,0.08069226,0.0001563914,0.0001004588,0.001852077,0.0003063382,0.00001885625,0.0003747642],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03849045,"threshold_uncertainty_score":0.2035593,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3246438670497029,"score_gpt":0.4228627877583974,"score_spread":0.09821892070869448,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}