{"id":"W4389523803","doi":"10.18653/v1/2023.findings-emnlp.983","title":"GTA: Gated Toxicity Avoidance for LM Performance Preservation","year":2023,"lang":"en","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea","keywords":"Offensive; Perplexity; Computer science; Consistency (knowledge bases); Artificial intelligence; Grammar; Generative grammar; Natural language processing; Machine learning; Language model; Operations research; Engineering; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003028376,0.00007535196,0.00007243144,0.00007834646,0.0001919791,0.00009302158,0.0003545582,0.00004687248,0.00001163823],"category_scores_gemma":[0.0000621997,0.00006789175,0.00003553804,0.0007080673,0.0000111982,0.0008022347,0.000082269,0.00005908539,0.0002607015],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002383807,"about_ca_system_score_gemma":0.00002746309,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001658223,"about_ca_topic_score_gemma":0.00001399525,"domain_scores_codex":[0.9991969,0.00001694176,0.0001299703,0.0002499817,0.0001568779,0.0002492905],"domain_scores_gemma":[0.9994774,0.00005312368,0.00003971478,0.0002913104,0.00009218389,0.00004625287],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001336622,0.000189002,0.006123502,0.0003951332,0.0000754805,0.00001124913,0.001744413,0.009481326,0.1523088,0.07552322,0.1192427,0.6347715],"study_design_scores_gemma":[0.0002709753,0.0001440007,0.009100647,0.0000136564,0.00000169405,0.000002078568,0.00001004763,0.7745191,0.1842347,0.001565516,0.02999265,0.0001449359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5332953,0.000005850819,0.4619852,0.0008840265,0.0003666812,0.000236148,0.000001238816,0.0008865215,0.002338957],"genre_scores_gemma":[0.968119,0.00002693256,0.01757367,0.0002744221,0.00009231857,0.00006112259,0.000009561729,0.000008164932,0.01383477],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7650378,"threshold_uncertainty_score":0.3350877,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02641855170120785,"score_gpt":0.2470416496224299,"score_spread":0.2206230979212221,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}