{"id":"W2831165","doi":"10.55016/ojs/ajer.v55i4.55342","title":"The Consequential Validity of Student Ratings: What do Instructors Really Think?","year":2010,"lang":"en","type":"article","venue":"Alberta Journal of Educational Research","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Psychology; Strengths and weaknesses; Applied psychology; Accountability; Medical education; Sample (material); Perception; Social psychology; Higher education; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1669258,0.0003608275,0.001024554,0.001947103,0.001265194,0.004380492,0.00125358,0.001401513,0.0009351327],"category_scores_gemma":[0.5504837,0.00047124,0.0008747269,0.001493378,0.005272551,0.003940017,0.002061818,0.002446232,0.0005222751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00243323,"about_ca_system_score_gemma":0.003515395,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002671207,"about_ca_topic_score_gemma":0.002289493,"domain_scores_codex":[0.786651,0.1398083,0.01596708,0.005329593,0.05027218,0.001971911],"domain_scores_gemma":[0.344231,0.4866134,0.05944405,0.01972962,0.08539481,0.004587085],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000691775,0.0003444884,0.5609078,0.001784411,0.0006532884,0.0001895087,0.1015412,0.0004621676,0.003576852,0.005733917,0.006115784,0.3179988],"study_design_scores_gemma":[0.0003724559,0.003431314,0.7357238,0.007803694,0.0009162828,0.001632404,0.1269657,0.009797117,0.01231954,0.02705304,0.07339127,0.0005933364],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.876169,0.005109899,0.05915412,0.02375604,0.001825665,0.0006518953,0.0002815266,0.0002585763,0.03279335],"genre_scores_gemma":[0.9869066,0.0008190364,0.009143546,0.001688189,0.000407701,0.0002322013,0.0001466386,0.00008698943,0.0005690847],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8330742,"threshold_uncertainty_score":0.8827986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2472252017628741,"score_gpt":0.5587972615610401,"score_spread":0.311572059798166,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}