{"id":"W2272017371","doi":"10.1016/j.amjsurg.2015.10.008","title":"Reliable assessment of operative performance","year":2015,"lang":"en","type":"article","venue":"The American Journal of Surgery","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":16,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University Health Centre","funders":"","keywords":"Generalizability theory; Reliability (semiconductor); Variance (accounting); Medicine; Variance components; Medical physics; Physical therapy; Psychology; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005316861,0.000797118,0.0007040426,0.00308355,0.0004870528,0.002099134,0.0009153608,0.00137795,0.002593782],"category_scores_gemma":[0.03131047,0.0003722871,0.0005034092,0.001349313,0.0006231643,0.00201913,0.001504946,0.0009492533,0.002587742],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004097944,"about_ca_system_score_gemma":0.0009985784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008682308,"about_ca_topic_score_gemma":0.00188439,"domain_scores_codex":[0.9919412,0.002195625,0.00126529,0.001085809,0.003068091,0.0004439834],"domain_scores_gemma":[0.9645148,0.01060953,0.006731079,0.004422731,0.01255413,0.001167735],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00147828,0.0002583956,0.7667669,0.0003222841,0.0002172828,0.0002348955,0.0004803347,0.00179971,0.01539224,0.0007884979,0.006946041,0.2053151],"study_design_scores_gemma":[0.00006853789,0.001182632,0.9430355,0.0003274986,0.000223274,0.001529375,0.0007510696,0.01706156,0.02284057,0.002507861,0.01031729,0.0001547914],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8472546,0.00528257,0.1055402,0.001262355,0.00095354,0.00038048,0.004478363,0.001316694,0.03353115],"genre_scores_gemma":[0.9793898,0.0005948181,0.01592064,0.0001985729,0.0002581727,0.0001442829,0.001428163,0.0001124305,0.001953109],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005316861,"threshold_uncertainty_score":0.02811861,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09641378202547783,"score_gpt":0.3733352164273037,"score_spread":0.2769214344018258,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}