{"id":"W4205575031","doi":"10.14293/s2199-1006.1.sor.2021.0003.v1","title":"Gender bias in student evaluation of teaching or a mirage?","year":2021,"lang":"en","type":"preprint","venue":"ScienceOpen Research","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mount Royal University","funders":"","keywords":"Outlier; Psychology; Statistics; Gender bias; Set (abstract data type); Social psychology; Demography; Mathematics; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01908195,0.0002917253,0.000535038,0.001118371,0.0006420023,0.00159356,0.0005939838,0.0005568532,0.005212178],"category_scores_gemma":[0.09842972,0.0001569578,0.0004082386,0.0008672749,0.001301119,0.001421189,0.001361994,0.0007398454,0.001086559],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007420806,"about_ca_system_score_gemma":0.0004029417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001522582,"about_ca_topic_score_gemma":0.002021584,"domain_scores_codex":[0.9805979,0.008649252,0.001622975,0.001739874,0.006533507,0.0008565345],"domain_scores_gemma":[0.9355791,0.03184756,0.01631183,0.005312351,0.009177834,0.00177144],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001770523,0.0002715591,0.832516,0.0003930278,0.0003075317,0.0002952124,0.03140672,0.0001129491,0.003000114,0.002306594,0.006187107,0.1214327],"study_design_scores_gemma":[0.00006241295,0.000805385,0.9622945,0.0003064705,0.00008311539,0.0006962509,0.02096552,0.0007435055,0.002776221,0.002253444,0.008936892,0.00007617036],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.983655,0.001070737,0.003224005,0.002087205,0.0004402028,0.00009687233,0.000324658,0.00005065827,0.009050712],"genre_scores_gemma":[0.9981233,0.000105106,0.0005074837,0.0004349386,0.00009337503,0.00005675041,0.00007909098,0.00001625413,0.000583675],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.980918,"threshold_uncertainty_score":0.1009162,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8890308265153848,"score_gpt":0.713796567568246,"score_spread":0.1752342589471387,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}