{"id":"W4410876142","doi":"10.1007/s10207-025-01066-4","title":"A Data-centric approach for safe and secure large language models against threatening and toxic content","year":2025,"lang":"en","type":"article","venue":"International Journal of Information Security","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université du Québec à Chicoutimi","funders":"","keywords":"Computer science; Cryptography; Computer security; Internet privacy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008597138,0.001379978,0.001800551,0.002227274,0.001631306,0.005565112,0.005926039,0.003115974,0.005837755],"category_scores_gemma":[0.03271286,0.00185783,0.002751546,0.001925505,0.002917703,0.01272886,0.01108814,0.006412003,0.004649574],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00233594,"about_ca_system_score_gemma":0.00514601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003622484,"about_ca_topic_score_gemma":0.0063213,"domain_scores_codex":[0.9902835,0.002620509,0.001078996,0.001697073,0.003624848,0.000695015],"domain_scores_gemma":[0.9636385,0.01145916,0.001599557,0.01816926,0.004431669,0.0007019184],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002224969,0.0007973667,0.005416274,0.0004871565,0.0007311603,0.0007140661,0.001302534,0.1935542,0.03636415,0.1781399,0.02885703,0.5514112],"study_design_scores_gemma":[0.00006275577,0.0001048531,0.000164824,0.00002839647,0.00008695419,0.0001172272,0.0001454294,0.8695604,0.01579242,0.1081394,0.005755058,0.000042174],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004075861,0.0001320075,0.9889464,0.0005539552,0.00008571849,0.0001494052,0.0003804264,0.00511118,0.0005650548],"genre_scores_gemma":[0.2725867,0.0002451552,0.7171371,0.001087521,0.0002649297,0.0006392918,0.002175756,0.001499139,0.004364323],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008597138,"threshold_uncertainty_score":0.04546654,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02298082640046019,"score_gpt":0.2729595508045289,"score_spread":0.2499787244040688,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}