{"id":"W4388798538","doi":"10.1007/s42087-023-00377-z","title":"Making Sense of Today’s Use of Student Evaluations of Teaching (SET)","year":2023,"lang":"en","type":"article","venue":"Human Arenas","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Operationalization; Set (abstract data type); Workload; Mathematics education; Test (biology); Psychology; Student engagement; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005419333,0.00007516838,0.0001996166,0.0002448489,0.0004178824,0.00003685051,0.0002148477,0.00005065004,0.0002889098],"category_scores_gemma":[0.003141836,0.00007805999,0.00008605082,0.0002625235,0.0001953256,0.000379193,0.0001055674,0.0001616257,0.00002373204],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000623242,"about_ca_system_score_gemma":0.0001351322,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001338494,"about_ca_topic_score_gemma":0.001365107,"domain_scores_codex":[0.9969252,0.001210142,0.0004714649,0.0001564579,0.001073698,0.0001630895],"domain_scores_gemma":[0.9976625,0.001146,0.0006376011,0.0002718858,0.0002468858,0.00003509272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.00002903336,0.0005012958,0.1616222,0.0001706227,0.0002478118,0.000006732159,0.4376447,0.006207232,0.02402289,0.3532883,0.009257157,0.007001949],"study_design_scores_gemma":[0.000995126,0.0003025791,0.8936962,0.0008270072,0.0003562543,0.000001233236,0.06627238,0.00110162,0.00230827,0.009637725,0.02407268,0.0004288871],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.986387,0.00002787136,0.00006565964,0.0006904398,0.0001412536,0.0002409291,0.00001054998,0.00006967661,0.01236658],"genre_scores_gemma":[0.9974055,0.00001250223,0.000973821,0.00002601687,0.00006099673,0.000006994399,0.000009427366,0.00001076638,0.001493983],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.732074,"threshold_uncertainty_score":0.3761298,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5861145869692092,"score_gpt":0.5875478068753649,"score_spread":0.001433219906155747,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}