{"id":"W4407979374","doi":"10.1186/s12909-025-06896-3","title":"Student evaluations of teaching do not reflect student learning: an observational study","year":2025,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Observational study; Medical education; Psychology; Mathematics education; Medicine; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005983758,0.0002970078,0.0005584501,0.001031198,0.001081551,0.00093328,0.0005374162,0.000621436,0.0009784708],"category_scores_gemma":[0.02522964,0.0004125426,0.0004394408,0.0008412593,0.001061747,0.0009977296,0.001147923,0.001307555,0.0004605076],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007363875,"about_ca_system_score_gemma":0.001398965,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002419063,"about_ca_topic_score_gemma":0.00424908,"domain_scores_codex":[0.995482,0.002132142,0.0004609501,0.0004623269,0.001081265,0.0003814452],"domain_scores_gemma":[0.9731852,0.01059433,0.007945814,0.002306572,0.003440313,0.002527787],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002207538,0.001492393,0.9869251,0.00002748685,0.00003283727,0.0001549274,0.005410859,0.00003510857,0.0003717006,0.0000261801,0.000180858,0.005121706],"study_design_scores_gemma":[0.00005120663,0.004809251,0.9840891,0.00003324528,0.00004500094,0.0005578395,0.008106031,0.0004743526,0.0006355419,0.00007601752,0.001091025,0.00003133756],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9995351,0.00002950123,0.0001374682,0.00001572716,0.000002908217,0.00004062931,0.00003909668,0.00000242041,0.0001971602],"genre_scores_gemma":[0.9992977,0.00004957281,0.0002619245,0.00003919763,0.00000987814,0.00007337567,0.00009716238,0.000003607071,0.0001676099],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9940162,"threshold_uncertainty_score":0.03164548,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3952213544656644,"score_gpt":0.640712764802941,"score_spread":0.2454914103372766,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}