{"id":"W4406450848","doi":"10.48550/arxiv.2501.08284","title":"AfriHate: A Multilingual Collection of Hate Speech and Abusive Language Datasets for African Languages","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European Commission; DeepMind; Universität Hamburg; International Development Research Centre; Rockefeller Foundation","keywords":"Computer science; Linguistics; Natural language processing; Political science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003622594,0.0002707629,0.0003987493,0.0002964084,0.0001483572,0.0000991436,0.0005998497,0.0002487259,0.000007670699],"category_scores_gemma":[0.0002548152,0.0002835879,0.0001174316,0.0003492764,0.00007131561,0.0001566418,0.0008033696,0.0003350125,0.000006882059],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006519504,"about_ca_system_score_gemma":0.0001785093,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001082601,"about_ca_topic_score_gemma":0.0003612201,"domain_scores_codex":[0.9982958,0.00008800305,0.0003475811,0.0007491738,0.0002053817,0.0003140848],"domain_scores_gemma":[0.9985003,0.0001824897,0.0002870061,0.000780099,0.0001621752,0.00008795524],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001134134,0.001302721,0.03240884,0.007724028,0.001889335,0.0008155248,0.06631232,0.0008862318,0.1136246,0.002166448,0.01415001,0.7575859],"study_design_scores_gemma":[0.003333984,0.0007201473,0.03438595,0.001405078,0.00035889,0.0001008651,0.004408815,0.02214798,0.9216933,0.001438412,0.00825052,0.001756053],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9767114,0.0007914115,0.01879821,0.0002531202,0.0008073138,0.0009586702,0.0008300163,0.0002268105,0.0006230689],"genre_scores_gemma":[0.979338,0.000144448,0.01852392,0.00009411076,0.0001546426,0.0001192578,0.0003160521,0.00001805861,0.001291484],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8080688,"threshold_uncertainty_score":0.9999616,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02078586480009493,"score_gpt":0.2966906162401926,"score_spread":0.2759047514400976,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}