{"id":"W4385571961","doi":"10.18653/v1/2023.acl-short.17","title":"A Weakly Supervised Classifier and Dataset of White Supremacist Language","year":2023,"lang":"en","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; Carnegie Mellon University; University of Pittsburgh","keywords":"Classifier (UML); Computer science; Artificial intelligence; Counterexample; Generalization; Speech recognition; Natural language processing; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001969286,0.00006383815,0.0000883521,0.0001003882,0.00007249194,0.00004662952,0.0002442747,0.00003353422,0.00006345822],"category_scores_gemma":[0.00001892409,0.0000531089,0.00001919782,0.0004243694,0.00002752161,0.0002168001,0.0002575117,0.00005071879,0.0001085896],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000004529345,"about_ca_system_score_gemma":0.00001543809,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006574087,"about_ca_topic_score_gemma":0.00004919174,"domain_scores_codex":[0.9993573,0.00002630884,0.0001238624,0.0002039156,0.000138467,0.0001501482],"domain_scores_gemma":[0.9995058,0.00002848106,0.00002415229,0.000362694,0.00001896214,0.00005998154],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006344158,0.0001685769,0.007780331,0.0003003819,0.0001149232,0.0003385317,0.01344969,0.00007042725,0.4510555,0.0213968,0.2497509,0.2555104],"study_design_scores_gemma":[0.002952068,0.0007408085,0.04532873,0.0001150527,0.00004459759,0.0002270674,0.003445923,0.5196712,0.2865187,0.002612686,0.1370035,0.00133967],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9062672,0.00006288929,0.08226492,0.001260472,0.0003544007,0.000226498,0.0002773501,0.0006594425,0.00862678],"genre_scores_gemma":[0.9826276,0.00002857188,0.01351686,0.0002197185,0.00003453422,0.000007228061,0.0001946896,0.000008219744,0.003362581],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5196008,"threshold_uncertainty_score":0.2165718,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01541823408194179,"score_gpt":0.2504659125987392,"score_spread":0.2350476785167974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}