{"id":"W4205575031","doi":"10.14293/s2199-1006.1.sor.2021.0003.v1","title":"Gender bias in student evaluation of teaching or a mirage?","year":2021,"lang":"en","type":"preprint","venue":"ScienceOpen Research","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mount Royal University","funders":"","keywords":"Outlier; Psychology; Statistics; Gender bias; Set (abstract data type); Social psychology; Demography; Mathematics; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","scholarly_communication","research_integrity","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3030994,0.0001740665,0.0003911918,0.001114146,0.001208554,0.001371433,0.002875952,0.0002894082,0.002894652],"category_scores_gemma":[0.06724693,0.0001522599,0.000102553,0.001685682,0.00103533,0.001001607,0.002724844,0.002823154,0.00005562875],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002005973,"about_ca_system_score_gemma":0.02339287,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.048387,"about_ca_topic_score_gemma":0.04914324,"domain_scores_codex":[0.9425707,0.03584903,0.0008867988,0.001283762,0.01843542,0.0009742649],"domain_scores_gemma":[0.9924837,0.003224151,0.0004490598,0.0008796075,0.002765229,0.0001982571],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001187309,0.001505633,0.103627,0.0003323267,0.0001178568,0.00006356702,0.7479343,0.02211163,0.001020876,0.02023783,0.002293024,0.1006372],"study_design_scores_gemma":[0.001731855,0.0002349817,0.3068917,0.001742576,0.0001347145,0.00000364415,0.6330881,0.02477927,0.0004030266,0.02003477,0.01001521,0.0009402229],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8537758,0.0004541088,0.00003533783,0.003981558,0.0006658153,0.002021984,0.000003377078,0.00002292638,0.1390391],"genre_scores_gemma":[0.9935588,0.0001837981,0.003366218,0.00004304069,0.0001555158,0.0002509953,0.000009625456,0.00001460552,0.002417404],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2358524,"threshold_uncertainty_score":0.9996653,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8890308265153848,"score_gpt":0.713796567568246,"score_spread":0.1752342589471387,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}