{"id":"W4396832273","doi":"10.1145/3613904.3642669","title":"The ``Colonial Impulse\" of Natural Language Processing: An Audit of Bengali Sentiment Analysis Tools and Their Identity-based Biases","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Defense Advanced Research Projects Agency","keywords":"Bengali; Sentiment analysis; Computer science; Sociotechnical system; Artificial intelligence; Colonialism; Identity (music); Natural language processing; Political science; Aesthetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01739845,0.000757264,0.000493493,0.006752968,0.002606006,0.004679508,0.001113479,0.0005340239,0.002158545],"category_scores_gemma":[0.06588906,0.0005966167,0.0005575178,0.008285143,0.003336461,0.003851922,0.004355073,0.001465034,0.002218],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00222689,"about_ca_system_score_gemma":0.003067122,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009653536,"about_ca_topic_score_gemma":0.0180572,"domain_scores_codex":[0.9825593,0.006302156,0.002294208,0.001488367,0.006584221,0.0007716914],"domain_scores_gemma":[0.9096071,0.04085637,0.00645411,0.01427713,0.02792442,0.0008807956],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001183977,0.000184881,0.1288829,0.005435705,0.0002643022,0.002753455,0.1777567,0.001284726,0.1015116,0.01069891,0.03974111,0.5303017],"study_design_scores_gemma":[0.00005657012,0.0003261632,0.3459696,0.001561825,0.0002639648,0.002351353,0.05715598,0.0129887,0.103186,0.006753165,0.4689556,0.0004310975],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8872232,0.002758715,0.05080445,0.004105393,0.0003464563,0.001511872,0.01168889,0.007259029,0.03430209],"genre_scores_gemma":[0.8794414,0.002311826,0.08764547,0.00104202,0.0001947681,0.001138744,0.01138594,0.005769695,0.01107023],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01739845,"threshold_uncertainty_score":0.09201288,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01767442416322,"score_gpt":0.2905672516684043,"score_spread":0.2728928275051843,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}