{"id":"W7106814607","doi":"10.48448/6fvp-e488","title":"Beneath the Facade: Probing Safety Vulnerabilities in LLMs via Auto-Generated Jailbreak Prompts","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Adversarial system; Resilience (materials science); Pace; Vulnerability (computing); Obstacle; Backdoor; Range (aeronautics); Trojan; Psychological resilience","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002745193,0.0008322705,0.0003344421,0.0009349991,0.000544159,0.001259372,0.0009525325,0.00120709,0.003566391],"category_scores_gemma":[0.016178,0.0003448481,0.0005846067,0.0003317502,0.001054376,0.002176905,0.002348717,0.00171266,0.001853358],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000528876,"about_ca_system_score_gemma":0.0006625275,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001072317,"about_ca_topic_score_gemma":0.002547926,"domain_scores_codex":[0.9976997,0.001180445,0.00009958696,0.0003688044,0.0005123001,0.0001392995],"domain_scores_gemma":[0.9916648,0.004618471,0.0006774911,0.002148469,0.0005835801,0.0003071004],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00252149,0.00132426,0.08619189,0.001755775,0.0003601729,0.002647819,0.004574842,0.3092759,0.07020652,0.03374923,0.1044981,0.382894],"study_design_scores_gemma":[0.0001650955,0.0008617079,0.01438916,0.0002710034,0.00008942345,0.0009275101,0.001082061,0.8579971,0.04089447,0.0345396,0.04860346,0.0001794184],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5435086,0.00154496,0.3594114,0.002719413,0.0008431776,0.0008839401,0.008502865,0.06436784,0.0182177],"genre_scores_gemma":[0.8885388,0.0002475062,0.0964668,0.0007421799,0.00008465228,0.000313613,0.007476832,0.0015702,0.004559416],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003566391,"threshold_uncertainty_score":0.01451814,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03402443923895525,"score_gpt":0.3050046783459499,"score_spread":0.2709802391069946,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}