{"id":"W7125604742","doi":"10.1109/cascon66301.2025.00020","title":"Enhancing Machine Learning in Abusive Language Detection with Dataset Integration","year":2025,"lang":"","type":"article","venue":"","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of Toronto; Vector Institute","funders":"","keywords":"Generalizability theory; Benchmark (surveying); Sampling (signal processing); Limiting; Language model; Co-occurrence; Core (optical fiber); Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006401541,0.0002712668,0.0002457494,0.0006194077,0.0003106719,0.0004014598,0.0003448873,0.0001465492,0.0001020387],"category_scores_gemma":[0.0001537756,0.0002350947,0.00003981798,0.001912037,0.00004445607,0.0008665681,0.0001677067,0.0007780056,0.00009009809],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002286024,"about_ca_system_score_gemma":0.000168483,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004732137,"about_ca_topic_score_gemma":0.04979731,"domain_scores_codex":[0.9979679,0.0002466351,0.0004203962,0.0006911039,0.0002655362,0.0004084432],"domain_scores_gemma":[0.9991395,0.00009473196,0.0001499469,0.0004552449,0.00009108378,0.00006944744],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001784374,0.0001154601,0.0004199983,0.00008210123,0.00004470194,0.0001043741,0.004951387,0.001860077,0.166894,0.0006313007,0.00006722834,0.8246509],"study_design_scores_gemma":[0.0008990556,0.0004618096,0.001066804,0.0004755716,0.00002732951,0.00003483181,0.002039212,0.3763446,0.6164777,0.00007254048,0.001768456,0.0003320787],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1169463,0.0002580927,0.8794385,0.0002519206,0.0005096287,0.0003482596,0.00001668667,0.0001640421,0.002066596],"genre_scores_gemma":[0.9918843,0.00007544956,0.004966958,0.0002799956,0.00005332343,0.00002383534,0.0001589978,0.00001151474,0.00254566],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.874938,"threshold_uncertainty_score":0.9675414,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004976048825501673,"score_gpt":0.2353654502951303,"score_spread":0.2303894014696286,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}