{"id":"W7082245607","doi":"10.48448/tmnx-pp69","title":"AfroBench: How Good are Large Language Models on African Languages?","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; University of Waterloo; McGill University","funders":"","keywords":"Discoverability; Languages of Africa; Scarcity; Representation (politics); Natural language; Natural language generation; Paraphrase","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008985543,0.003538828,0.001341647,0.002846062,0.001675923,0.003812206,0.003011464,0.002493795,0.01189617],"category_scores_gemma":[0.02922082,0.0006183114,0.001728863,0.002151116,0.0009158759,0.0111026,0.003961602,0.003333217,0.007494742],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001816456,"about_ca_system_score_gemma":0.002870093,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01999783,"about_ca_topic_score_gemma":0.02925169,"domain_scores_codex":[0.9944558,0.00290339,0.000356696,0.001281931,0.0005816065,0.0004206547],"domain_scores_gemma":[0.9901034,0.006030241,0.0003015173,0.001784537,0.001290206,0.0004901888],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003838636,0.001725954,0.0253612,0.003469605,0.00120624,0.0005609678,0.001447616,0.06580325,0.01227955,0.008666138,0.2330293,0.6426116],"study_design_scores_gemma":[0.001158192,0.001657602,0.01517157,0.001472475,0.0006699887,0.00073353,0.005065452,0.7399918,0.03424282,0.04283857,0.1566433,0.0003545954],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5688772,0.02307828,0.1504605,0.01657908,0.004634144,0.001782775,0.06027341,0.1016073,0.07270722],"genre_scores_gemma":[0.7501822,0.002368153,0.1426908,0.002697587,0.0003593947,0.0009795455,0.08665062,0.003214941,0.01085678],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01999783,"threshold_uncertainty_score":0.0475207,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01552435403433872,"score_gpt":0.2561214672024378,"score_spread":0.2405971131680991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}