{"id":"W4392669860","doi":"10.18653/v1/2023.ijcnlp-main.10","title":"MasakhaNEWS: News Topic Classification for African languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute; University of Waterloo","funders":"DeepMind","keywords":"Computer science; Languages of Africa; Natural language processing; Artificial intelligence; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002034987,0.002509126,0.001221877,0.01491246,0.001957609,0.005134067,0.001286443,0.001697748,0.01973927],"category_scores_gemma":[0.008265186,0.0006269535,0.002346735,0.008189401,0.0003621978,0.005893323,0.003485221,0.001898421,0.0265533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001091856,"about_ca_system_score_gemma":0.001811946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01153112,"about_ca_topic_score_gemma":0.01194101,"domain_scores_codex":[0.9981919,0.0003274662,0.0002294049,0.0004460968,0.0005564446,0.0002487153],"domain_scores_gemma":[0.9968464,0.001230386,0.0003245188,0.0005098353,0.0004920101,0.0005969299],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009881731,0.0002610769,0.01307301,0.002225965,0.0003564894,0.0007639017,0.0007663619,0.001766585,0.007667304,0.002838421,0.6624807,0.306812],"study_design_scores_gemma":[0.0003944312,0.0004925366,0.03735074,0.0009521857,0.0004174941,0.001830504,0.004170689,0.101685,0.01405207,0.008163033,0.830255,0.0002361682],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.06695849,0.01717774,0.08317236,0.006095121,0.004375651,0.002269327,0.6017626,0.1780352,0.04015353],"genre_scores_gemma":[0.07270505,0.004893589,0.1255179,0.0007107373,0.001065168,0.001428248,0.7653911,0.003503249,0.02478485],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01973927,"threshold_uncertainty_score":0.06603444,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0562600828557314,"score_gpt":0.3205969545643245,"score_spread":0.264336871708593,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}