{"id":"W3151539628","doi":"10.1016/j.amjsurg.2021.03.064","title":"Objective Structured Assessment of technical skill (OSATS) in the Surgical Skills and Technology Elective Program (SSTEP): Comparison of peer and expert raters","year":2021,"lang":"en","type":"article","venue":"The American Journal of Surgery","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":36,"is_retracted":false,"has_abstract":false,"ca_institutions":"Kingston Health Sciences Centre; Kingston General Hospital; Queen's University","funders":"","keywords":"Intraclass correlation; Reliability (semiconductor); Medical education; Task (project management); Medicine; Session (web analytics); Medical physics; Psychology; Physical therapy; Computer science; Psychometrics; Clinical psychology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001053444,0.000109658,0.0008257181,0.0002126953,0.00003552252,0.00001056022,0.00007315522,0.00005185483,0.00001620206],"category_scores_gemma":[0.0004458669,0.0000583252,0.0001119426,0.0008501239,0.0008175697,0.00003896037,0.00003756617,0.000458891,4.63827e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003787553,"about_ca_system_score_gemma":0.0001832683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000122592,"about_ca_topic_score_gemma":0.000004059632,"domain_scores_codex":[0.9983755,0.000339107,0.0005847153,0.000125549,0.0004142387,0.0001608639],"domain_scores_gemma":[0.9967521,0.002088797,0.0005603529,0.0001398427,0.0003985439,0.00006035703],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006017734,0.0008383732,0.8097104,0.00001862481,0.0002377467,0.0002713723,0.002242926,0.00004132489,0.001827857,0.0002542786,0.0000580437,0.1838972],"study_design_scores_gemma":[0.001525412,0.000648491,0.9780653,0.0001929868,0.00007874263,0.001262891,0.01443328,0.0001674233,0.002774015,0.0002194618,0.0005434427,0.00008859173],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9964596,0.0004700515,0.00003508154,0.002580734,0.00004401361,0.0001457514,0.00000166616,0.000007559457,0.0002555178],"genre_scores_gemma":[0.998928,0.0001673938,0.0007151132,0.0001292,0.00003959613,0.000005354905,0.000001747631,0.000008040874,0.000005535783],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1838086,"threshold_uncertainty_score":0.301237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0229720155041902,"score_gpt":0.3806514651017053,"score_spread":0.3576794495975151,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}