{"id":"W6891511957","doi":"10.48448/qbjb-af66","title":"AfriInstruct: Instruction Tuning of African Languages for Diverse Tasks","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Task (project management); Baseline (sea); Languages of Africa; Language model; Stability (learning theory); Language acquisition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000767345,0.0003858752,0.0004664323,0.002272442,0.0001392556,0.0001266416,0.0009305189,0.0002253922,0.0006532563],"category_scores_gemma":[0.0003023589,0.0003517741,0.0001290655,0.002059651,0.002097751,0.0002428842,0.0003377734,0.0002948663,0.000803689],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002977976,"about_ca_system_score_gemma":0.0005387316,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006842196,"about_ca_topic_score_gemma":0.0008459558,"domain_scores_codex":[0.9971601,0.00003061076,0.0003919863,0.0008966174,0.0009353429,0.0005853018],"domain_scores_gemma":[0.9984242,0.00004393395,0.0005089562,0.0006564691,0.0001991637,0.0001672157],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001250277,0.0002508808,0.00034254,0.002182887,0.0004497008,0.00006112939,0.00472332,0.0001521112,0.07867067,0.09315803,0.671987,0.1478967],"study_design_scores_gemma":[0.002084395,0.0004741852,0.00009384163,0.001880956,0.0006083566,0.000103362,0.02192505,0.01094588,0.005784659,0.006426289,0.947913,0.001760055],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.003624623,0.001386052,0.002224333,0.0001527824,0.003594529,0.001653356,0.004110703,0.001558756,0.9816949],"genre_scores_gemma":[0.4868045,0.00005057055,0.0962515,0.00006389432,0.001933507,0.0001016319,0.0003363812,0.002374337,0.4120836],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5696113,"threshold_uncertainty_score":0.9999743,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02813317264915929,"score_gpt":0.3187061539515721,"score_spread":0.2905729813024128,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}