{"id":"W4397001364","doi":"10.21203/rs.3.rs-4224772/v1","title":"Student Evaluations of Teaching Do Not Reflect Student Learning: An Observational Study","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Observational study; Observational learning; Mathematics education; Psychology; Pedagogy; Computer science; Experiential learning; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01276179,0.0003710783,0.0006501309,0.001348127,0.001813735,0.001708118,0.0008677494,0.0009579221,0.001397984],"category_scores_gemma":[0.08659608,0.000586691,0.0005037652,0.0009584439,0.001875179,0.001562744,0.001913707,0.002129161,0.0007441995],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009611113,"about_ca_system_score_gemma":0.001965302,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002127995,"about_ca_topic_score_gemma":0.003308846,"domain_scores_codex":[0.9886279,0.007063756,0.0008292414,0.0007135171,0.002162767,0.0006029771],"domain_scores_gemma":[0.8644007,0.09047635,0.01763899,0.009718115,0.01190339,0.00586247],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001510379,0.008183342,0.8167163,0.0001337801,0.0001215347,0.0005900337,0.1363204,0.0002521812,0.003811686,0.0004454628,0.0006852524,0.03122974],"study_design_scores_gemma":[0.0002464674,0.01217167,0.8976213,0.00008884988,0.0001470476,0.001262801,0.07445049,0.001396709,0.007383518,0.0005879733,0.004513256,0.000129898],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9990845,0.00001800079,0.0002857827,0.00002386182,0.00000372903,0.0000429202,0.0000308547,0.000004041186,0.0005063533],"genre_scores_gemma":[0.9989151,0.00003306721,0.0003140434,0.00002811779,0.000007175484,0.0001068756,0.0000499622,0.000007858235,0.0005378309],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9872382,"threshold_uncertainty_score":0.06749159,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6985254627341579,"score_gpt":0.699522918917968,"score_spread":0.000997456183810086,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}