{"id":"W7102431493","doi":"","title":"AfriMTEB and AfriE5: Benchmarking and Adapting Text Embedding Models for African Languages","year":2025,"lang":"","type":"article","venue":"arXiv (Cornell University)","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Canada First Research Excellence Fund","keywords":"Embedding; Benchmarking; Component (thermodynamics); Adaptation (eye); Languages of Africa; Cluster analysis","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004962017,0.003849147,0.001113704,0.00310812,0.001487578,0.002189931,0.002977599,0.00271516,0.008268106],"category_scores_gemma":[0.01309603,0.0006579496,0.002045513,0.001874606,0.001117149,0.004799891,0.00390684,0.003757605,0.008958159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001461387,"about_ca_system_score_gemma":0.001392137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0136823,"about_ca_topic_score_gemma":0.02429576,"domain_scores_codex":[0.9974135,0.001080872,0.0001953304,0.0007771085,0.0002983408,0.0002348975],"domain_scores_gemma":[0.9968171,0.001338076,0.0001141562,0.0009469069,0.0005819743,0.0002017949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001768847,0.001418965,0.01157958,0.002123561,0.001018087,0.0007790488,0.0009788632,0.1044335,0.01577384,0.004651711,0.2047346,0.6507394],"study_design_scores_gemma":[0.0007181301,0.001590802,0.008233532,0.0005179506,0.000369947,0.001209173,0.001747809,0.8227309,0.03345041,0.01113026,0.1179777,0.0003234794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5862476,0.01975144,0.1856266,0.004651898,0.00703551,0.002205318,0.06215687,0.08886047,0.04346441],"genre_scores_gemma":[0.5399014,0.002484697,0.2168078,0.002213278,0.0005859394,0.001428155,0.2077943,0.00458402,0.02420034],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0136823,"threshold_uncertainty_score":0.0276596,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03903952360992188,"score_gpt":0.1995726577564995,"score_spread":0.1605331341465776,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}