{"id":"W4321392937","doi":"10.48550/arxiv.2302.08956","title":"AfriSenti: A Twitter Sentiment Analysis Benchmark for African Languages","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"DeepMind; Universität Hamburg; International Development Research Centre; Rockefeller Foundation","keywords":"Amharic; Computer science; Arabic languages; Natural language processing; Task (project management); Languages of Africa; Artificial intelligence; Portuguese; Annotation; Benchmark (surveying); Linguistics; Arabic; World Wide Web; Geography","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001560835,0.001413476,0.0004682642,0.003683679,0.001914063,0.001151454,0.0008755232,0.001041871,0.003787796],"category_scores_gemma":[0.00524043,0.0002061792,0.0006607569,0.002710998,0.0003365937,0.002027185,0.001730215,0.0008325895,0.003415862],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009862018,"about_ca_system_score_gemma":0.0008421363,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009234749,"about_ca_topic_score_gemma":0.01908526,"domain_scores_codex":[0.9986473,0.0004081361,0.0001793091,0.0002068335,0.0003969596,0.0001615871],"domain_scores_gemma":[0.9982008,0.0005544745,0.0001988607,0.00024458,0.0006085301,0.0001927575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001593751,0.000752977,0.03124155,0.00316697,0.000324846,0.0009572536,0.002840386,0.00693979,0.03104144,0.005370643,0.7500877,0.1656827],"study_design_scores_gemma":[0.0003969555,0.0005742002,0.1244156,0.0005456691,0.0001718353,0.001189251,0.007280905,0.1198548,0.03233981,0.007840401,0.7051427,0.000247853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.3858828,0.002971387,0.01690552,0.003962753,0.001607143,0.001696632,0.5165141,0.01602886,0.05443099],"genre_scores_gemma":[0.2648707,0.0008588477,0.0449332,0.0008045139,0.0003801266,0.001670969,0.6752862,0.001066006,0.01012934],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.009234749,"threshold_uncertainty_score":0.01836199,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1100176813714993,"score_gpt":0.2375084513938683,"score_spread":0.127490770022369,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}