{"id":"W4389523860","doi":"10.18653/v1/2023.emnlp-main.862","title":"AfriSenti: A Twitter Sentiment Analysis Benchmark for African Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Linguistics and Language Analysis","field":"Arts and Humanities","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"DeepMind; Universität Hamburg; International Development Research Centre; Rockefeller Foundation","keywords":"Benchmark (surveying); Art; Artificial intelligence; Humanities; Computer science; Art history; Cartography; Geography","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001892784,0.001315021,0.0006068806,0.003562132,0.001805264,0.001320268,0.0008178494,0.00109716,0.004174736],"category_scores_gemma":[0.00538784,0.0002634211,0.0006416747,0.002486937,0.0002953981,0.003075478,0.002169095,0.0007040352,0.005400081],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005973117,"about_ca_system_score_gemma":0.0008182342,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005796088,"about_ca_topic_score_gemma":0.01094012,"domain_scores_codex":[0.9987319,0.0003823232,0.0001691802,0.0001655225,0.000375435,0.0001756704],"domain_scores_gemma":[0.9982604,0.0004623387,0.0001645533,0.0002061505,0.0006750182,0.000231523],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00243827,0.0008636268,0.03723859,0.002531088,0.0003956431,0.0007088556,0.001890996,0.00652737,0.0252495,0.005899308,0.6492691,0.2669876],"study_design_scores_gemma":[0.0006755156,0.0009531523,0.1134134,0.0005465664,0.0002591234,0.001167196,0.007844362,0.1986309,0.03085044,0.01087124,0.6344942,0.0002938567],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.5885559,0.006514272,0.02294216,0.00568621,0.002567275,0.002096665,0.2844634,0.01896803,0.06820616],"genre_scores_gemma":[0.5221985,0.001988646,0.05158257,0.000960791,0.0008261536,0.002145542,0.3954865,0.001556209,0.02325511],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.005796088,"threshold_uncertainty_score":0.0139659,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03223315102030863,"score_gpt":0.2847308466247436,"score_spread":0.252497695604435,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}