{"id":"W4287890642","doi":"10.18653/v1/2022.gebnlp-1.17","title":"Choose Your Lenses: Flaws in Gender Bias Evaluation","year":2022,"lang":"en","type":"article","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation","keywords":"Metric (unit); Computer science; Measure (data warehouse); Task (project management); Gender bias; Position (finance); Affect (linguistics); Machine learning; Artificial intelligence; Econometrics; Data mining; Psychology; Social psychology; Mathematics; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1607471,0.001422643,0.001223466,0.004714949,0.003139078,0.005428964,0.003020043,0.002590924,0.005222075],"category_scores_gemma":[0.4930391,0.0006429066,0.001065228,0.003347168,0.005168854,0.009048348,0.007142735,0.003313766,0.001632266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00346472,"about_ca_system_score_gemma":0.003048832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00312459,"about_ca_topic_score_gemma":0.004024521,"domain_scores_codex":[0.7684609,0.1595799,0.01117427,0.009696955,0.04921935,0.001868668],"domain_scores_gemma":[0.5394853,0.3211573,0.02443081,0.04903005,0.06231194,0.003584631],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002124081,0.0004171156,0.1139969,0.002353789,0.0009244155,0.0004468472,0.01764199,0.004667383,0.008543639,0.1332766,0.07079296,0.6448143],"study_design_scores_gemma":[0.0006205937,0.001937992,0.109764,0.005273914,0.0005771039,0.00252386,0.01935563,0.04532357,0.04119524,0.5378016,0.2348177,0.00080888],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2654155,0.0106136,0.5797219,0.05110924,0.005278673,0.001711283,0.003507176,0.00405063,0.07859205],"genre_scores_gemma":[0.8066726,0.001133207,0.1705665,0.0101072,0.0008504366,0.001835001,0.001311399,0.00196749,0.005556265],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8392529,"threshold_uncertainty_score":0.850122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.466993520118383,"score_gpt":0.4856308287517183,"score_spread":0.01863730863333529,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}