{"id":"W4285211372","doi":"10.18653/v1/2022.acl-long.378","title":"Improving Generalizability in Implicitly Abusive Language Detection with Concept Activation Vectors","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Generalizability theory; Computer science; Robustness (evolution); Interpretability; Artificial intelligence; Moderation; Machine learning; Natural language processing; Metric (unit); Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0251356,0.002837686,0.002428673,0.003156891,0.001132671,0.003370167,0.002891725,0.003267299,0.001189202],"category_scores_gemma":[0.09239104,0.00103761,0.001479997,0.001837989,0.003306033,0.006336321,0.004958706,0.007373793,0.0006740881],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001288333,"about_ca_system_score_gemma":0.001246811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005776135,"about_ca_topic_score_gemma":0.005100007,"domain_scores_codex":[0.9887294,0.006178727,0.0007668269,0.002699414,0.001167435,0.0004580865],"domain_scores_gemma":[0.9228069,0.05995898,0.003753203,0.009137569,0.003697161,0.0006461818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001233114,0.0005909979,0.05103995,0.000411575,0.0007081408,0.0006450205,0.002414624,0.4579304,0.02031248,0.009266149,0.003814713,0.4516328],"study_design_scores_gemma":[0.00002164303,0.0001685748,0.003182428,0.00003967762,0.00004306189,0.0001126929,0.0001579196,0.9784681,0.00414655,0.01313526,0.0004895265,0.00003462496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2948201,0.001378222,0.6969259,0.001378897,0.0001796608,0.0003325266,0.0003178967,0.002502376,0.002164415],"genre_scores_gemma":[0.9375101,0.000307634,0.05957261,0.0003894511,0.000121162,0.0001697116,0.0006708579,0.000219121,0.001039303],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0251356,"threshold_uncertainty_score":0.1329313,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.003906633246131122,"score_gpt":0.2045248411760247,"score_spread":0.2006182079298936,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}