{"id":"W4385894685","doi":"10.18653/v1/2022.blackboxnlp-1.18","title":"Towards Procedural Fairness: Uncovering Biases in How a Toxic Language Classifier Uses Sentiment Information","year":2022,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Classifier (UML); Debiasing; Salient; Leverage (statistics); Artificial intelligence; Sentiment analysis; Machine learning; Natural language processing; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02680799,0.0009219609,0.001238593,0.001087753,0.001285472,0.003619738,0.001667954,0.002054901,0.001704673],"category_scores_gemma":[0.09195284,0.0004726368,0.0007592482,0.0006436252,0.00437837,0.005571072,0.003575697,0.004013353,0.0004246922],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00188336,"about_ca_system_score_gemma":0.001998428,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002071414,"about_ca_topic_score_gemma":0.001982913,"domain_scores_codex":[0.9901143,0.005628941,0.0003279081,0.001478382,0.001728827,0.0007216212],"domain_scores_gemma":[0.9399279,0.04125261,0.005714054,0.009244831,0.002837399,0.001023351],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002701494,0.0006461618,0.1105163,0.0003699862,0.0006692185,0.0006429558,0.003689619,0.4413311,0.02633016,0.2255198,0.0057657,0.1818175],"study_design_scores_gemma":[0.00003931952,0.0001569247,0.004585919,0.0000456868,0.00004699194,0.0001039604,0.0002278546,0.8262696,0.008086191,0.1594278,0.0009604763,0.00004924352],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4053566,0.000314322,0.5844009,0.002753712,0.0001085919,0.0001718652,0.0001520059,0.0004271907,0.006314792],"genre_scores_gemma":[0.9765385,0.00005266924,0.02179505,0.0004898023,0.00003684796,0.000046146,0.00007863313,0.00007023807,0.0008921261],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02680799,"threshold_uncertainty_score":0.1417759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01775743042092704,"score_gpt":0.2606138412506929,"score_spread":0.2428564108297658,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}