{"id":"W7125604742","doi":"10.1109/cascon66301.2025.00020","title":"Enhancing Machine Learning in Abusive Language Detection with Dataset Integration","year":2025,"lang":"","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of Toronto; Vector Institute","funders":"","keywords":"Generalizability theory; Benchmark (surveying); Sampling (signal processing); Limiting; Language model; Co-occurrence; Core (optical fiber); Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0140335,0.002493366,0.001400504,0.002937963,0.001152484,0.002789116,0.002273833,0.002412016,0.0009150931],"category_scores_gemma":[0.03819451,0.0005467565,0.001991355,0.002527178,0.001119829,0.004393481,0.004813356,0.003825428,0.001096773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008540306,"about_ca_system_score_gemma":0.00132338,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00531877,"about_ca_topic_score_gemma":0.009545855,"domain_scores_codex":[0.9898191,0.006318725,0.0006209604,0.002055724,0.0008405488,0.0003448599],"domain_scores_gemma":[0.9821066,0.00922516,0.001197223,0.005149763,0.001913416,0.0004079081],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001232152,0.00290593,0.2522117,0.001099228,0.002287445,0.0004980568,0.001398504,0.2448649,0.01364871,0.003759681,0.02573706,0.4503567],"study_design_scores_gemma":[0.0001566822,0.0008427537,0.03500247,0.0002277105,0.0004942953,0.0003960786,0.0007288541,0.9246467,0.0100629,0.01457126,0.01270033,0.000170015],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6889797,0.005081356,0.275477,0.004261454,0.0007448274,0.00110052,0.00845201,0.008309032,0.007593998],"genre_scores_gemma":[0.8423322,0.0004946698,0.1318628,0.001537706,0.0002345334,0.0006950794,0.02106738,0.0002772345,0.00149834],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0140335,"threshold_uncertainty_score":0.07421708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004976048825501673,"score_gpt":0.2353654502951303,"score_spread":0.2303894014696286,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}