{"id":"W3161895874","doi":"10.1080/0142159x.2021.1925642","title":"On the validity of summative entrustment decisions","year":2021,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Toronto; University of Ottawa; Medical Council of Canada","funders":"","keywords":"Summative assessment; Argument (complex analysis); Psychology; Process (computing); Health care; External validity; Nursing; Medical education; Formative assessment; Medicine; Social psychology; Computer science; Pedagogy; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009390453,0.0000773659,0.000248824,0.00001476037,0.00003579162,0.000004287914,0.00008719679,0.0001144669,0.02152835],"category_scores_gemma":[0.346549,0.00004017069,0.0001445831,0.0001283895,0.0001603343,0.000005972292,0.00006429872,0.0003835686,0.0001748536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003032352,"about_ca_system_score_gemma":0.0002619567,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001113107,"about_ca_topic_score_gemma":0.000003482153,"domain_scores_codex":[0.9983736,0.000213246,0.0002713043,0.0001690869,0.0008298162,0.0001429511],"domain_scores_gemma":[0.9765694,0.02251691,0.00005855943,0.0004198531,0.0001088366,0.0003264357],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0002946612,0.009801793,0.189479,0.00003071501,0.0005250868,0.001428081,0.001699897,0.000004344891,0.0001018788,0.0712819,0.5900898,0.1352628],"study_design_scores_gemma":[0.01988909,0.004175238,0.6642712,0.02812418,0.001583351,0.0004259612,0.004248822,0.007360335,0.01915875,0.09488693,0.1548505,0.001025638],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9437332,0.0001429891,0.0004337053,0.02480692,0.0002118019,0.0001226737,0.00000333659,0.00002134856,0.03052402],"genre_scores_gemma":[0.9926665,0.0001319884,0.0001334278,0.002649103,0.0001495619,0.00001327758,0.0000159018,0.000007925175,0.004232311],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4747922,"threshold_uncertainty_score":0.9793661,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09308446346101831,"score_gpt":0.3871435513230113,"score_spread":0.294059087861993,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}