{"id":"W7082245607","doi":"10.48448/tmnx-pp69","title":"AfroBench: How Good are Large Language Models on African Languages?","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; University of Waterloo; McGill University","funders":"","keywords":"Discoverability; Languages of Africa; Scarcity; Representation (politics); Natural language; Natural language generation; Paraphrase","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006430808,0.0004063234,0.0003922787,0.0005843739,0.0002628872,0.0003828529,0.003235745,0.0002730417,0.0003287413],"category_scores_gemma":[0.0002958244,0.0003494057,0.00009237981,0.001405041,0.0002959897,0.0002672643,0.00098542,0.0004693745,0.0001038387],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009731088,"about_ca_system_score_gemma":0.0004798854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001316023,"about_ca_topic_score_gemma":0.0002109462,"domain_scores_codex":[0.9968973,0.00005612343,0.0001939104,0.001225022,0.0007933821,0.0008341977],"domain_scores_gemma":[0.9976771,0.00008394212,0.0003174756,0.00158357,0.0001418303,0.000196037],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000005087019,0.000446746,0.00006948862,0.0002768532,0.00006626375,0.0004073557,0.001508363,0.0008897562,0.0008270813,0.6015916,0.3760591,0.0178523],"study_design_scores_gemma":[0.0008771603,0.000164006,0.00007612241,0.001056328,0.00003788603,0.0000305176,0.004028216,0.163508,0.002516135,0.01423513,0.8117942,0.00167634],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00001156224,0.0004120136,0.1030348,0.005952298,0.0004152918,0.0002284252,0.00008975907,0.0006091261,0.8892467],"genre_scores_gemma":[0.1636351,0.00002724157,0.01340675,0.001099462,0.0002880967,0.0000206613,0.0000334584,0.00003233555,0.8214569],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.5873564,"threshold_uncertainty_score":0.9998958,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01552435403433872,"score_gpt":0.2561214672024378,"score_spread":0.2405971131680991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}