{"id":"W3122321999","doi":"10.14293/s2199-1006.1.sor.2021.0001.v1","title":"Small samples, unreasonable generalizations, and outliers: Gender bias in student evaluation of teaching or three unhappy students?","year":2021,"lang":"en","type":"preprint","venue":"ScienceOpen Research","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mount Royal University","funders":"","keywords":"Psychology; Outlier; Set (abstract data type); Gender bias; Sample (material); Social psychology; Section (typography); Statistics; Mathematics; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.03443596,0.0001935734,0.0003851018,0.0009307674,0.0005486072,0.0005858231,0.00230366,0.0002534186,0.003582368],"category_scores_gemma":[0.002041539,0.0001764456,0.00005154476,0.001159678,0.0004243423,0.0001319034,0.003174271,0.001545042,0.00002339254],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006490055,"about_ca_system_score_gemma":0.003823191,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.01784557,"about_ca_topic_score_gemma":0.02323328,"domain_scores_codex":[0.9891216,0.005562442,0.0008117001,0.001005075,0.002914058,0.0005851576],"domain_scores_gemma":[0.995002,0.0008087369,0.0002690059,0.001792717,0.001920867,0.0002066948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004089626,0.001132231,0.8896098,0.0001818159,0.00006815814,0.000003436906,0.05962354,0.0005497721,0.00003437574,0.01360203,0.001543373,0.03361063],"study_design_scores_gemma":[0.0007395071,0.00005994596,0.8976866,0.0002952429,0.00002343273,0.000004322377,0.09481575,0.001759751,0.000007635155,0.003296538,0.001115378,0.000195907],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9755069,0.004803924,0.0003586521,0.001845849,0.0007064466,0.002567305,0.0000223,0.00001900794,0.01416958],"genre_scores_gemma":[0.9912371,0.0006675658,0.004493944,0.0001272228,0.00007522615,0.001006184,0.0001402171,0.00002431551,0.002228248],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03519221,"threshold_uncertainty_score":0.9973285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8522916635758395,"score_gpt":0.6483096562453999,"score_spread":0.2039820073304396,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}