{"id":"W2599980125","doi":"10.1080/2331186x.2017.1304016","title":"Student evaluations of teaching are an inadequate assessment tool for evaluating faculty performance","year":2017,"lang":"en","type":"article","venue":"Cogent Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":307,"is_retracted":false,"has_abstract":true,"ca_institutions":"Algoma University","funders":"","keywords":"Summative assessment; Set (abstract data type); Medical education; Formative assessment; Faculty development; Psychology; Evaluation methods; Program evaluation; Course evaluation; Teaching method; Mathematics education; Higher education; Pedagogy; Computer science; Medicine; Professional development; Engineering; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06190298,0.0008124847,0.001813917,0.006243567,0.001677656,0.005329505,0.002105394,0.001079584,0.002362876],"category_scores_gemma":[0.2482231,0.0003589364,0.001069837,0.006746664,0.002180524,0.004138407,0.002268608,0.002154504,0.001526823],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002904999,"about_ca_system_score_gemma":0.003478527,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002840501,"about_ca_topic_score_gemma":0.01076409,"domain_scores_codex":[0.9039693,0.03778071,0.007912596,0.001608735,0.04784588,0.0008827209],"domain_scores_gemma":[0.5961902,0.2767953,0.04399134,0.01057116,0.06862002,0.003831955],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005593165,0.0004919603,0.225293,0.003599691,0.0005401167,0.00009251585,0.003441637,0.001102457,0.0007637182,0.004786192,0.0119228,0.7474065],"study_design_scores_gemma":[0.0001549817,0.008508162,0.8106585,0.01460977,0.001378784,0.001847111,0.04130065,0.01728038,0.007560553,0.04831675,0.04786387,0.0005204817],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7766247,0.0398821,0.0716836,0.0123926,0.002858296,0.0007562201,0.001764361,0.001582734,0.09245533],"genre_scores_gemma":[0.9570981,0.00688714,0.02875974,0.001401391,0.0003570353,0.0003802448,0.000497916,0.0001455682,0.004472987],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.938097,"threshold_uncertainty_score":0.3273782,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.330727264537629,"score_gpt":0.6184685834139221,"score_spread":0.2877413188762931,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}