{"id":"W1995874702","doi":"10.1080/13803611.2014.997915","title":"What response rates are needed to make reliable inferences from student evaluations of teaching?","year":2014,"lang":"en","type":"article","venue":"Educational Research and Evaluation","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Set (abstract data type); Range (aeronautics); Statistics; Confidence interval; Stability (learning theory); Psychology; Econometrics; Computer science; Mathematics; Engineering; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.367773,0.0008418611,0.002305447,0.001664913,0.001317604,0.004238761,0.002776375,0.00483109,0.002118395],"category_scores_gemma":[0.7539178,0.000918699,0.002377563,0.002468792,0.00269523,0.00683918,0.002301475,0.003774483,0.001573841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002946391,"about_ca_system_score_gemma":0.003662569,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002871785,"about_ca_topic_score_gemma":0.001740261,"domain_scores_codex":[0.4178891,0.5003746,0.02445764,0.01051914,0.04351627,0.003243255],"domain_scores_gemma":[0.2048143,0.6690598,0.0433725,0.04520369,0.03574501,0.001804796],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006159716,0.001693598,0.3247601,0.005384089,0.001984183,0.000482076,0.01716404,0.03424264,0.006637427,0.03513168,0.0193029,0.5470575],"study_design_scores_gemma":[0.002280499,0.0135481,0.5786208,0.008110695,0.001237076,0.001808119,0.01534936,0.1340463,0.03506852,0.1488616,0.05998214,0.001086697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.500298,0.002079512,0.4415056,0.02900873,0.00151869,0.004884685,0.003192856,0.0009382907,0.01657373],"genre_scores_gemma":[0.8907754,0.0003984043,0.09914783,0.002700478,0.0002335716,0.005488389,0.0006797266,0.0001472163,0.0004288563],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6322271,"threshold_uncertainty_score":0.7796485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3841725397005639,"score_gpt":0.6224553468616918,"score_spread":0.2382828071611279,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}