{"id":"W4385569921","doi":"10.18653/v1/2023.woah-1.21","title":"Factoring Hate Speech: A New Annotation Framework to Study Hate Speech in Social Media","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto; Hebrew University of Jerusalem","keywords":"Annotation; Computer science; Scheme (mathematics); Construct (python library); Social media; Natural language processing; Artificial intelligence; Voice activity detection; Speech recognition; Speech processing; World Wide Web; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005922408,0.00157793,0.0009415568,0.008378646,0.00250342,0.002577694,0.001404666,0.001678846,0.003139348],"category_scores_gemma":[0.01542029,0.0005611984,0.0008527791,0.004221856,0.002168088,0.005427013,0.003886852,0.002754338,0.002350942],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001240027,"about_ca_system_score_gemma":0.002176671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007558492,"about_ca_topic_score_gemma":0.01196009,"domain_scores_codex":[0.9947789,0.001851907,0.0004290687,0.001666701,0.0009500047,0.0003234337],"domain_scores_gemma":[0.9790468,0.009183172,0.0026596,0.003368877,0.004745561,0.0009959539],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001312558,0.0007357548,0.1000867,0.001773621,0.0003115637,0.0006025622,0.01712125,0.007250457,0.1191488,0.03799747,0.03124291,0.6824166],"study_design_scores_gemma":[0.0001635377,0.0009448283,0.235513,0.0007256143,0.0004908986,0.001947571,0.01125749,0.3200943,0.0712138,0.08868294,0.2680924,0.0008737534],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1108021,0.00157612,0.8556125,0.002004497,0.0007618212,0.0009806178,0.009997424,0.003906771,0.0143582],"genre_scores_gemma":[0.4506378,0.0008740235,0.5193595,0.0005803235,0.001064784,0.001896988,0.01424054,0.0009077656,0.0104383],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008378646,"threshold_uncertainty_score":0.03132105,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07390689382459951,"score_gpt":0.320097330515426,"score_spread":0.2461904366908265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}