{"id":"W4287890642","doi":"10.18653/v1/2022.gebnlp-1.17","title":"Choose Your Lenses: Flaws in Gender Bias Evaluation","year":2022,"lang":"en","type":"article","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Azrieli Foundation","keywords":"Metric (unit); Computer science; Measure (data warehouse); Task (project management); Gender bias; Position (finance); Affect (linguistics); Machine learning; Artificial intelligence; Econometrics; Data mining; Psychology; Social psychology; Mathematics; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00603077,0.0000505359,0.00008135938,0.00007485887,0.0008351613,0.00009634213,0.0001748102,0.00005994951,0.004591382],"category_scores_gemma":[0.001250512,0.0000524738,0.00004380669,0.0003577642,0.0000750031,0.0002533791,0.00006425935,0.0002600789,0.00003224028],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003368809,"about_ca_system_score_gemma":0.0006566577,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02044861,"about_ca_topic_score_gemma":0.01330125,"domain_scores_codex":[0.9977835,0.0007494452,0.0001361843,0.0001383896,0.0009563674,0.0002361062],"domain_scores_gemma":[0.9995046,0.0001380171,0.00004605197,0.0000928091,0.000146375,0.00007217377],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002334956,0.0003042389,0.02246276,0.00000673043,0.00002737639,0.00001163463,0.387561,0.0006577577,0.0001943525,0.528457,0.0217284,0.03856536],"study_design_scores_gemma":[0.001207165,0.0001162838,0.06155617,0.000005916927,0.00002987817,7.696231e-7,0.3157723,0.001322722,0.000055381,0.3674126,0.2520605,0.0004603048],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5633973,0.0001161218,0.000005891222,0.009878451,0.0003449341,0.0002495259,0.000002695198,0.00003664453,0.4259684],"genre_scores_gemma":[0.9935836,0.00006171816,0.00006285228,0.00212455,0.0001379085,0.00003251432,0.000004488115,0.000006038801,0.003986326],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4301863,"threshold_uncertainty_score":0.9963186,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.466993520118383,"score_gpt":0.4856308287517183,"score_spread":0.01863730863333529,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}