{"id":"W4411120192","doi":"10.18653/v1/2025.naacl-long.302","title":"Stronger Universal and Transferable Attacks by Suppressing Refusals","year":2025,"lang":"en","type":"article","venue":"","topic":"Cryptography and Data Security","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Tamkeen; Army Research Office; King Abdulaziz City for Science and Technology; York University; New York University Abu Dhabi; U.S. Department of Homeland Security; Open Philanthropy Project; Noyce Foundation; National Science Foundation","keywords":"Computer science; Computer security","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01474033,0.002137868,0.003972491,0.002828928,0.004018562,0.005828687,0.004711133,0.004990396,0.01332285],"category_scores_gemma":[0.0732796,0.002005856,0.002661938,0.002485205,0.009084033,0.02340324,0.01760745,0.01030472,0.004937245],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002024074,"about_ca_system_score_gemma":0.002989106,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007243152,"about_ca_topic_score_gemma":0.0009171003,"domain_scores_codex":[0.9742955,0.009158534,0.001761496,0.003715602,0.007387117,0.00368188],"domain_scores_gemma":[0.8923731,0.05863837,0.004204961,0.03761961,0.005354705,0.001809281],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001212485,0.0004787739,0.001568006,0.0005881116,0.000268639,0.0005897896,0.002049237,0.01718948,0.0158829,0.818717,0.02610696,0.1153485],"study_design_scores_gemma":[0.0002093622,0.0001282334,0.0002200712,0.00009063839,0.0001462243,0.0003111524,0.0001864174,0.05484726,0.009506416,0.9215702,0.01269966,0.00008439987],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1495822,0.002062008,0.7202749,0.01526121,0.002222961,0.0007187038,0.0008299705,0.0105508,0.09849723],"genre_scores_gemma":[0.9158571,0.0007323087,0.05935998,0.003705473,0.001723809,0.0003709809,0.0004768035,0.001586071,0.01618744],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01474033,"threshold_uncertainty_score":0.07795525,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.003495829172001487,"score_gpt":0.2244669642449901,"score_spread":0.2209711350729886,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}