{"id":"W4397001364","doi":"10.21203/rs.3.rs-4224772/v1","title":"Student Evaluations of Teaching Do Not Reflect Student Learning: An Observational Study","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Observational study; Observational learning; Mathematics education; Psychology; Pedagogy; Computer science; Experiential learning; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","scholarly_communication","research_integrity"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.08656919,0.0003166917,0.0005172815,0.001029737,0.002471054,0.00154755,0.002049927,0.0003392728,0.0007978658],"category_scores_gemma":[0.01877098,0.0003166222,0.0002289499,0.0007433929,0.0003670815,0.0004977038,0.003155244,0.006908653,0.0002099541],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001523795,"about_ca_system_score_gemma":0.004147384,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006744604,"about_ca_topic_score_gemma":0.004138682,"domain_scores_codex":[0.9531446,0.02997403,0.001068549,0.001318941,0.01372042,0.0007734402],"domain_scores_gemma":[0.9909529,0.004414175,0.000521933,0.001022992,0.00277932,0.0003086781],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.00009479528,0.005239902,0.3751121,0.000465126,0.00069487,0.00002822582,0.5102663,0.01772722,0.000243563,0.07707553,0.001590678,0.01146162],"study_design_scores_gemma":[0.0006539873,0.001841043,0.5827496,0.001037243,0.0003107903,6.322609e-7,0.3774414,0.001268598,0.0000199179,0.01574899,0.01835323,0.0005745323],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9534814,0.0006164102,0.0000296512,0.008777592,0.000796055,0.003988229,0.0000356182,0.0002607638,0.03201431],"genre_scores_gemma":[0.9912255,0.0001391424,0.0007981634,0.00002503708,0.0009515026,0.0007180413,0.0001022573,0.0000657942,0.005974512],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2076375,"threshold_uncertainty_score":0.9999286,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6985254627341579,"score_gpt":0.699522918917968,"score_spread":0.000997456183810086,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}