{"id":"W4311192871","doi":"10.36227/techrxiv.21708188","title":"Regional language toxic comment classification","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Marathi; Popularity; Social media; Hindi; Computer science; Gujarati; Focus (optics); Entertainment; Artificial intelligence; Natural language processing; Data science; World Wide Web; Political science; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005834756,0.0009892436,0.0005012163,0.00404261,0.000866137,0.00115621,0.0008173952,0.0007757073,0.01607681],"category_scores_gemma":[0.002728999,0.000122963,0.001062417,0.002954461,0.0002891237,0.0007469367,0.0009531743,0.0006823982,0.01576628],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009579865,"about_ca_system_score_gemma":0.001122808,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02577064,"about_ca_topic_score_gemma":0.04225338,"domain_scores_codex":[0.9989395,0.0001795789,0.000104027,0.0002843987,0.0002386975,0.000253846],"domain_scores_gemma":[0.9978047,0.0003953596,0.0002326541,0.0003707155,0.0009751399,0.0002214856],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001078083,0.0004892579,0.158465,0.00189836,0.0002723739,0.001914986,0.001386019,0.004828287,0.01940067,0.002215318,0.4885155,0.319536],"study_design_scores_gemma":[0.0001195694,0.0004735815,0.3201034,0.0004472339,0.0003176999,0.002075692,0.007719635,0.08070162,0.02899383,0.001844011,0.5570114,0.0001923593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5464017,0.003567769,0.009216323,0.001988585,0.001347129,0.0006906986,0.362785,0.008810636,0.06519214],"genre_scores_gemma":[0.5344009,0.0008007861,0.01294834,0.000409123,0.0004415708,0.0004655123,0.4077182,0.0004563653,0.04235915],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02577064,"threshold_uncertainty_score":0.05378228,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03997279582722367,"score_gpt":0.2830121337848864,"score_spread":0.2430393379576627,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}