{"id":"W4393228430","doi":"10.1002/ail2.92","title":"Building Text and Speech Benchmark Datasets and Models for Low‐Resourced East African Languages: Experiences and Lessons","year":2024,"lang":"en","type":"article","venue":"Applied AI Letters","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"FP7 International Cooperation; Deutsche Gesellschaft für Internationale Zusammenarbeit; International Development Research Centre; Rockefeller Foundation","keywords":"Benchmark (surveying); Computer science; Linguistics; Natural language processing; Artificial intelligence; Geography; Cartography","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01078842,0.001934305,0.0008807623,0.001811899,0.00180868,0.002288539,0.002837817,0.002261675,0.002505786],"category_scores_gemma":[0.01745291,0.0005229386,0.001218925,0.002286032,0.001225077,0.004754424,0.002682783,0.002830225,0.002612752],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002281867,"about_ca_system_score_gemma":0.00200959,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03631575,"about_ca_topic_score_gemma":0.03723905,"domain_scores_codex":[0.9948692,0.003167514,0.0003504635,0.0006869127,0.0005748289,0.0003510278],"domain_scores_gemma":[0.9860364,0.008059096,0.0003518895,0.002180954,0.002577965,0.0007936477],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002295705,0.006112504,0.06145982,0.001937347,0.0007319697,0.002133312,0.003742801,0.247737,0.02476841,0.01132687,0.1955211,0.4422332],"study_design_scores_gemma":[0.0004596282,0.001087195,0.02786207,0.0004631795,0.0001864225,0.0004844423,0.006376214,0.8367409,0.03579244,0.0123659,0.07794166,0.0002400783],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8309215,0.002671502,0.08031941,0.01331668,0.001234269,0.001348744,0.04547391,0.0146693,0.01004477],"genre_scores_gemma":[0.7306937,0.001143827,0.1094272,0.0009365462,0.0003357755,0.001149892,0.1501156,0.0009773356,0.005219928],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03631575,"threshold_uncertainty_score":0.0722087,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01178532502583053,"score_gpt":0.2778462994356153,"score_spread":0.2660609744097848,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}