{"id":"W7125393895","doi":"10.55492/v6i02.6747","title":"Building Corpora for Low-Resource Kenyan Languages","year":2025,"lang":"","type":"article","venue":"Journal of the Digital Humanities Association of Southern Africa","topic":"ICT in Developing Communities","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Bundesministerium für Wirtschaftliche Zusammenarbeit und Entwicklung; International Development Research Centre; Rockefeller Foundation","keywords":"Languages of Africa; Kenya; Reading (process); Natural language; Corpus linguistics; Text corpus; Computational linguistics; Language identification","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001532186,0.0003002681,0.0006392379,0.0004210882,0.0006431197,0.001265092,0.003026466,0.0001826191,0.00001613402],"category_scores_gemma":[0.001947637,0.0002454715,0.0007171411,0.0005165608,0.0002205472,0.0006523583,0.0008581609,0.0005219555,0.000007792351],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008389162,"about_ca_system_score_gemma":0.0005240757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009086844,"about_ca_topic_score_gemma":0.000006075382,"domain_scores_codex":[0.996769,0.0002607053,0.001389026,0.0001614335,0.0009720242,0.0004478833],"domain_scores_gemma":[0.9918259,0.001924566,0.004172947,0.0005612618,0.001470106,0.00004524743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006594753,0.002031297,0.02002244,0.002792856,0.006697657,0.00002041399,0.380153,0.002993115,0.0007536015,0.3623522,0.1587299,0.06279404],"study_design_scores_gemma":[0.005401439,0.001087475,0.004104871,0.01009551,0.0008639129,0.00004136064,0.1608914,0.004083615,0.005275569,0.1592274,0.6472645,0.001662953],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5760401,0.01033549,0.1337848,0.02804658,0.01293577,0.002770582,0.003488878,0.0002889784,0.2323087],"genre_scores_gemma":[0.9534128,0.00001760986,0.001277307,0.0003445792,0.0003189144,0.000004775421,0.000002341995,0.00003162707,0.04459009],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4885346,"threshold_uncertainty_score":0.9999998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01795061024483665,"score_gpt":0.2419945099350392,"score_spread":0.2240438996902026,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}