{"id":"W4415318169","doi":"10.48550/arxiv.2510.07203","title":"Sunflower: A New Approach To Expanding Coverage of African Languages in Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"International Development Research Centre","keywords":"Swahili; Comprehension; Languages of Africa; Bantu languages; State (computer science); Language model","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006103674,0.0004142893,0.000695142,0.0007304787,0.00005242214,0.0001363211,0.0027007,0.0003563147,0.00001140346],"category_scores_gemma":[0.000222803,0.0003998641,0.0001704174,0.0009885357,0.00002726125,0.0003691426,0.003682989,0.0009096313,0.000009375957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002005082,"about_ca_system_score_gemma":0.0003761042,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001196429,"about_ca_topic_score_gemma":0.0001233898,"domain_scores_codex":[0.997301,0.000127459,0.0005421086,0.001063784,0.0004107177,0.0005549232],"domain_scores_gemma":[0.9978305,0.0001148632,0.0002648138,0.001543897,0.00008957336,0.0001563681],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004546758,0.004213406,0.1403168,0.0100101,0.000832606,0.001550365,0.3529898,0.02123769,0.03450319,0.2281762,0.01660357,0.1891116],"study_design_scores_gemma":[0.009469087,0.000704903,0.02772206,0.02048039,0.0004723256,0.0001213402,0.01105272,0.2450441,0.4328521,0.2345352,0.003844473,0.01370125],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1605001,0.004645386,0.8200029,0.0003527396,0.0002197998,0.0007190551,0.00006446398,0.0006797601,0.01281581],"genre_scores_gemma":[0.7847899,0.00004802979,0.2128635,0.0003844634,0.00007938979,0.00006788418,0.00002997079,0.0000221771,0.001714763],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6242898,"threshold_uncertainty_score":0.9998453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03171535518445631,"score_gpt":0.3121366835627905,"score_spread":0.2804213283783342,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}