{"id":"W4412862721","doi":"10.5120/ijca2025925482","title":"Semantic Jailbreaks and RLHF Limitations in LLMs: A Taxonomy, Failure Trace, and Mitigation Strategy","year":2025,"lang":"en","type":"article","venue":"International Journal of Computer Applications","topic":"Digital and Cyber Forensics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trinity College","funders":"","keywords":"Computer science; TRACE (psycholinguistics); Taxonomy (biology); Data science; Ecology; Linguistics; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00703691,0.001180907,0.0006357527,0.001930477,0.001061771,0.001766431,0.001904145,0.001967244,0.004719522],"category_scores_gemma":[0.05006614,0.0004416933,0.0004595426,0.0007159376,0.002549967,0.005619786,0.002913562,0.002752095,0.001040792],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001066437,"about_ca_system_score_gemma":0.001145572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001526924,"about_ca_topic_score_gemma":0.001981561,"domain_scores_codex":[0.9924616,0.002606998,0.0005808616,0.0008255071,0.0029263,0.0005987347],"domain_scores_gemma":[0.9433827,0.03138762,0.006790732,0.01335687,0.004219363,0.0008626303],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004493253,0.001671778,0.06301667,0.002324946,0.0003115808,0.003462507,0.009741638,0.09964384,0.1034752,0.05195586,0.01799943,0.6419033],"study_design_scores_gemma":[0.0002190128,0.004186279,0.03041362,0.001515455,0.0004220001,0.00787356,0.008494805,0.6128194,0.2221298,0.07230158,0.03905709,0.0005673685],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6519008,0.001267124,0.3135659,0.00262975,0.0002025246,0.0007677117,0.0006853549,0.01203657,0.01694434],"genre_scores_gemma":[0.9638401,0.0001459776,0.0334956,0.0003035717,0.00001952761,0.0001196305,0.00022967,0.0002257505,0.001620074],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00703691,"threshold_uncertainty_score":0.03721517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01773420956557482,"score_gpt":0.2427508446670002,"score_spread":0.2250166351014254,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}