{"id":"W3132011002","doi":"10.48550/arxiv.2105.05541","title":"Evaluating Gender Bias in Natural Language Inference","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Debiasing; Gender bias; Task (project management); Premise; Inference; Computer science; Natural (archaeology); Natural language processing; Artificial intelligence; Psychology; Social psychology; Cognitive psychology; Linguistics; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.00058903,0.0003411293,0.0003663332,0.0004079135,0.00008491803,0.0003351555,0.002338642,0.0003249182,0.00002946137],"category_scores_gemma":[0.000463345,0.0003741003,0.0001543499,0.001047861,0.00006601131,0.0006739972,0.003802723,0.001412372,0.00001394422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003322045,"about_ca_system_score_gemma":0.0004505413,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004372057,"about_ca_topic_score_gemma":0.0002126564,"domain_scores_codex":[0.997562,0.0003224349,0.000251973,0.001258821,0.0001885588,0.0004162321],"domain_scores_gemma":[0.9979397,0.0002196947,0.0002661036,0.001248309,0.00023345,0.00009274184],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001400087,0.0008702351,0.0468803,0.002394706,0.0004744144,0.02612257,0.03390803,0.1552764,0.01548161,0.5984761,0.0003574487,0.1196182],"study_design_scores_gemma":[0.0004239957,0.00003091695,0.001326089,0.0004926323,0.00003289808,0.00001691459,0.0004433513,0.9291057,0.003293965,0.0639964,0.000006846837,0.0008303194],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6023248,0.003629522,0.392266,0.00005132275,0.0004516235,0.0002368021,0.000003197953,0.0006302351,0.0004064843],"genre_scores_gemma":[0.8772027,0.00005477208,0.1221838,0.0001458711,0.00004139134,0.000001522718,0.00002085486,0.00001573756,0.0003333647],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7738292,"threshold_uncertainty_score":0.9998711,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1913847993683349,"score_gpt":0.2933840641489097,"score_spread":0.1019992647805749,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}