{"id":"W4311204652","doi":"10.36227/techrxiv.21708188.v1","title":"Regional language toxic comment classification","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Marathi; Popularity; Hindi; Social media; Focus (optics); Computer science; Gujarati; Entertainment; Artificial intelligence; Natural language processing; Data science; World Wide Web; Political science; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000676592,0.001028295,0.0005307594,0.003679797,0.0008613273,0.001112863,0.0007795451,0.0008041139,0.01165022],"category_scores_gemma":[0.003050189,0.0001133316,0.001045044,0.00245766,0.0002890266,0.0007498207,0.0009500906,0.0006929913,0.01056203],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007964093,"about_ca_system_score_gemma":0.001046831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01768451,"about_ca_topic_score_gemma":0.03020467,"domain_scores_codex":[0.9988844,0.0001920617,0.0001145423,0.0002829614,0.0002671983,0.0002588002],"domain_scores_gemma":[0.9974923,0.0004625911,0.0002686595,0.0003816142,0.00115367,0.0002412146],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001495403,0.0006014347,0.2140708,0.002238344,0.0003877017,0.002469062,0.001530832,0.006715756,0.02358489,0.002486839,0.3930806,0.3513383],"study_design_scores_gemma":[0.0001412718,0.0005696517,0.3421782,0.0005220568,0.000423843,0.002663305,0.00845576,0.1099575,0.03428672,0.002248363,0.4983357,0.0002176837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6571279,0.004196917,0.01247574,0.001830442,0.001525868,0.000706133,0.2607333,0.007370818,0.05403284],"genre_scores_gemma":[0.6391057,0.0008923964,0.01589881,0.0003728575,0.0005380672,0.0004706421,0.3050719,0.0004204562,0.03722917],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01768451,"threshold_uncertainty_score":0.03897381,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03997279582722367,"score_gpt":0.2830121337848864,"score_spread":0.2430393379576627,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}