{"id":"W4312067146","doi":"10.1097/acm.0000000000005006","title":"Necessary but Insufficient and Possibly Counterproductive: The Complex Problem of Teaching Evaluations","year":2022,"lang":"en","type":"article","venue":"Academic Medicine","topic":"Innovations in Medical Education","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Sunnybrook Health Science Centre; The Wilson Centre; Sinai Health System","funders":"","keywords":"Formative assessment; Construct (python library); Psychology; Attractiveness; Subject (documents); Portfolio; Appeal; Mathematics education; Medical education; Social psychology; Computer science; Medicine; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3363166,0.001416275,0.003108389,0.006747053,0.004166244,0.02188564,0.006595593,0.01533853,0.002452572],"category_scores_gemma":[0.7263429,0.001393221,0.001688445,0.004947944,0.03667426,0.03020294,0.00796569,0.01654903,0.001429886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01187715,"about_ca_system_score_gemma":0.01502566,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004374824,"about_ca_topic_score_gemma":0.005505587,"domain_scores_codex":[0.4008455,0.4701167,0.03719005,0.01801393,0.07093664,0.0028971],"domain_scores_gemma":[0.1436493,0.7433702,0.02356504,0.02434205,0.06247725,0.002596295],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002780139,0.00006028614,0.003199517,0.004425289,0.0003204244,0.0003860298,0.02567104,0.0004564295,0.0002587495,0.4730547,0.178596,0.3132935],"study_design_scores_gemma":[0.0001465819,0.0002111462,0.003235198,0.01695516,0.000269518,0.001083392,0.01321487,0.00211891,0.001224338,0.5406449,0.4205392,0.0003568283],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.005718247,0.0521729,0.05083846,0.8448246,0.01929442,0.0002434472,0.0001790781,0.0003548692,0.02637393],"genre_scores_gemma":[0.3937041,0.0320528,0.1242176,0.3932156,0.04322623,0.002191484,0.0002072869,0.001454941,0.009729976],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6636834,"threshold_uncertainty_score":0.8184398,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04383531746988024,"score_gpt":0.3977613292119577,"score_spread":0.3539260117420774,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}