{"id":"W2608565195","doi":"10.1007/s00464-017-5569-y","title":"Establishing meaningful benchmarks: the development of a formative feedback tool for advanced laparoscopic suturing","year":2017,"lang":"en","type":"article","venue":"Surgical Endoscopy","topic":"Surgical Simulation and Training","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; McGill University Health Centre","funders":"","keywords":"Formative assessment; Likert scale; Medicine; Medical physics; Task (project management); Psychology; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006114069,0.0001659648,0.0004094795,0.00004494459,0.0006138182,0.00009447309,0.0002390274,0.00007817156,0.0002974021],"category_scores_gemma":[0.0009139365,0.0001066284,0.000149513,0.00007286712,0.0001319272,0.0003232164,0.0001024978,0.0001996031,0.000007067037],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000537166,"about_ca_system_score_gemma":0.0001236457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006026334,"about_ca_topic_score_gemma":0.000009186096,"domain_scores_codex":[0.9985779,0.00003261289,0.0004928277,0.0002286299,0.0003291026,0.0003389489],"domain_scores_gemma":[0.9981352,0.0009441468,0.0002903294,0.0003785489,0.0001385432,0.0001132309],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.02897835,0.001794911,0.1357394,0.003016537,0.002060449,0.0002257921,0.06648443,0.002370293,0.004794351,0.05258022,0.0008314092,0.7011238],"study_design_scores_gemma":[0.2042507,0.0009653961,0.101462,0.003175186,0.000337898,0.00005022516,0.00646182,0.008823151,0.1738668,0.001988216,0.4973625,0.001256082],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9710974,0.00007959026,0.000498766,0.0003400304,0.0003707818,0.00057003,0.000007004184,0.00003669251,0.02699968],"genre_scores_gemma":[0.9922159,0.000009689836,0.00719367,0.00007068081,0.0001629296,0.00007183045,0.00003605266,0.00001646227,0.0002228409],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6998678,"threshold_uncertainty_score":0.4721056,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03242681642580231,"score_gpt":0.3328721437959679,"score_spread":0.3004453273701656,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}