{"id":"W4385569921","doi":"10.18653/v1/2023.woah-1.21","title":"Factoring Hate Speech: A New Annotation Framework to Study Hate Speech in Social Media","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto; Hebrew University of Jerusalem","keywords":"Annotation; Computer science; Scheme (mathematics); Construct (python library); Social media; Natural language processing; Artificial intelligence; Voice activity detection; Speech recognition; Speech processing; World Wide Web; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0009922932,0.0005084198,0.0005952651,0.000856371,0.0002170688,0.0009393617,0.001610115,0.0005371877,0.00006412244],"category_scores_gemma":[0.0004872433,0.0005318089,0.0001691388,0.001665985,0.00001693399,0.0003527925,0.002140581,0.00141919,0.0009489085],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003058998,"about_ca_system_score_gemma":0.0002381664,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002493875,"about_ca_topic_score_gemma":0.002182921,"domain_scores_codex":[0.9958485,0.0002082058,0.0007199566,0.001467137,0.001022478,0.0007337164],"domain_scores_gemma":[0.9980841,0.0002762354,0.0002374516,0.0009571732,0.0001450397,0.000299964],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000070854,0.0004131065,0.003433523,0.00008993825,0.0001792315,0.0008409046,0.08813469,0.002111615,0.0005220593,0.002721051,0.002403568,0.8990794],"study_design_scores_gemma":[0.004312622,0.00118661,0.4274349,0.001908819,0.0001907606,0.00005386053,0.00866633,0.02124515,0.04960934,0.4735906,0.004750251,0.007050816],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5596853,0.0000209328,0.4262011,0.002980732,0.007222604,0.001657822,0.00000842406,0.001567965,0.0006551516],"genre_scores_gemma":[0.8306151,0.00002570056,0.1647016,0.0002922946,0.001980597,0.0001435729,0.00002338359,0.00009283884,0.002124847],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8920286,"threshold_uncertainty_score":0.999829,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07390689382459951,"score_gpt":0.320097330515426,"score_spread":0.2461904366908265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}