{"id":"W4385569777","doi":"10.18653/v1/2023.woah-1.14","title":"Concept-Based Explanations to Test for False Causal Relationships Learned by Abusive Language Classifiers","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; Artificial intelligence; Set (abstract data type); Feature (linguistics); Machine learning; Focus (optics); Natural language processing; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03381132,0.001807429,0.001201388,0.003025094,0.001121487,0.002399257,0.001923541,0.003330625,0.003611232],"category_scores_gemma":[0.1607063,0.0003769427,0.001261101,0.001721695,0.002736553,0.006108958,0.003150021,0.005891737,0.0006984463],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001056152,"about_ca_system_score_gemma":0.001235533,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00139029,"about_ca_topic_score_gemma":0.001666283,"domain_scores_codex":[0.9830442,0.008579748,0.001447605,0.002830446,0.003323938,0.0007740569],"domain_scores_gemma":[0.6735955,0.2886958,0.01263827,0.0143966,0.009008816,0.001664963],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005156829,0.001841569,0.5194976,0.001846905,0.002378047,0.001168356,0.005216203,0.09796228,0.01165731,0.028797,0.01904866,0.3054292],"study_design_scores_gemma":[0.0004394922,0.002298187,0.1150578,0.0005307166,0.0009023936,0.001613699,0.003103815,0.7536263,0.02980974,0.07925034,0.01308277,0.0002847445],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7951541,0.001964068,0.1893397,0.00226478,0.0003894651,0.0007397573,0.002740648,0.001672132,0.005735273],"genre_scores_gemma":[0.9643657,0.0001088343,0.03194629,0.0003293819,0.00009163341,0.0003083727,0.002218993,0.0001060784,0.000524831],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03381132,"threshold_uncertainty_score":0.1788135,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07954067804392582,"score_gpt":0.3169069890658719,"score_spread":0.2373663110219461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}