{"id":"W6929293509","doi":"10.48448/v6jj-sw89","title":"AfriSenti: A Twitter Sentiment Analysis Benchmark for African Languages","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Animal Behavior and Reproduction","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Annotation; Benchmark (surveying); Baseline (sea); Task (project management); Languages of Africa; Diversity (politics); Lexical diversity","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001574338,0.001722551,0.0004370745,0.003645687,0.001669598,0.00119587,0.001001514,0.001191146,0.005533262],"category_scores_gemma":[0.005243503,0.000220393,0.0007081652,0.002365469,0.0003485996,0.002061273,0.002020642,0.0008789812,0.005645513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009931447,"about_ca_system_score_gemma":0.0008657859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009284161,"about_ca_topic_score_gemma":0.02390042,"domain_scores_codex":[0.9986781,0.000426504,0.0001480952,0.0002224344,0.0003700877,0.000154688],"domain_scores_gemma":[0.9981646,0.0006045539,0.0002034632,0.0002679197,0.0005548772,0.0002046066],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001239715,0.0006023715,0.02257771,0.002807127,0.0002262261,0.0007514652,0.002267426,0.00485061,0.01703997,0.00435915,0.8134352,0.1298431],"study_design_scores_gemma":[0.0002763186,0.0003728147,0.09049474,0.0005177041,0.0001190552,0.000967497,0.004988057,0.07169164,0.02137258,0.007369509,0.8016046,0.0002253492],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2037333,0.002886855,0.01655758,0.0034614,0.001475021,0.001536227,0.6954115,0.01985958,0.05507841],"genre_scores_gemma":[0.13577,0.0006422417,0.03646269,0.0007170376,0.0002861938,0.001458195,0.8130315,0.001100479,0.01053158],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009284161,"threshold_uncertainty_score":0.01851058,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05325854834247259,"score_gpt":0.3198733142519082,"score_spread":0.2666147659094357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}