{"id":"W4412889667","doi":"10.18653/v1/2025.acl-long.1572","title":"Where Are We? Evaluating LLM Performance on African Languages","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Bill and Melinda Gates Foundation","keywords":"Computer science; Natural language processing; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001749413,0.0005442146,0.000463489,0.001079717,0.0007099136,0.001160464,0.0006546526,0.0008329398,0.003499118],"category_scores_gemma":[0.006642885,0.0001655418,0.0002879932,0.001316578,0.0003065746,0.002652542,0.001461235,0.0005380656,0.002386043],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005601032,"about_ca_system_score_gemma":0.0005244671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007694088,"about_ca_topic_score_gemma":0.009972519,"domain_scores_codex":[0.99899,0.0005283796,0.00007399702,0.0001334114,0.0001322551,0.0001419685],"domain_scores_gemma":[0.9975139,0.001461246,0.0001213127,0.0002266399,0.0004225719,0.0002542799],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006468008,0.00094468,0.0966348,0.001514614,0.0003073874,0.0007465523,0.003611928,0.0123012,0.0339275,0.003860324,0.0691418,0.7705411],"study_design_scores_gemma":[0.001074513,0.004963915,0.2485564,0.001308437,0.000715173,0.001614641,0.03616785,0.3986831,0.1393652,0.01333455,0.1538807,0.0003355847],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9746476,0.002790644,0.004303932,0.00197407,0.0001978751,0.00006894143,0.002940794,0.002316843,0.01075932],"genre_scores_gemma":[0.9798788,0.0006704686,0.01050472,0.0001658581,0.00004704792,0.0000439676,0.005708483,0.0001712196,0.002809392],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007694088,"threshold_uncertainty_score":0.0152986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02144773171065271,"score_gpt":0.333771858516156,"score_spread":0.3123241268055033,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}