{"id":"W4416017657","doi":"10.1145/3746252.3760929","title":"Uncovering the Persuasive Fingerprint of LLMs in Jailbreaking Attacks","year":2025,"lang":"","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Persuasion; Adversarial system; Readability; Fingerprint (computing); Interrogative; Sophistication","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001635879,0.000356185,0.0005149694,0.0003766142,0.000345396,0.0002255054,0.002429914,0.0001782054,0.0002049519],"category_scores_gemma":[0.000954026,0.0002842574,0.0002152928,0.00208709,0.0002787757,0.0004829105,0.0030909,0.001053816,0.00001734852],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000264967,"about_ca_system_score_gemma":0.0004313476,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001526439,"about_ca_topic_score_gemma":0.0001904794,"domain_scores_codex":[0.9967436,0.0004406106,0.0009067646,0.0007869626,0.0004964551,0.0006256141],"domain_scores_gemma":[0.9969629,0.001223325,0.0003961023,0.001197598,0.0001663122,0.00005377252],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000446709,0.0001017737,0.02366614,0.0001082765,0.0001064007,0.00003248053,0.01038697,0.6301481,0.0003550848,0.1150872,0.00004438524,0.2199184],"study_design_scores_gemma":[0.00111944,0.00007984877,0.03947887,0.0009688686,0.00004811861,0.000008796771,0.002196174,0.9446758,0.003019469,0.007002659,0.0009691254,0.0004328447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1175094,0.0005099186,0.8381358,0.003377812,0.001828257,0.0004030956,9.465895e-7,0.00005751183,0.03817724],"genre_scores_gemma":[0.9827892,0.00005106674,0.01569222,0.0004910103,0.00007140489,0.00001087638,3.821973e-7,0.00001533264,0.0008785639],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8652797,"threshold_uncertainty_score":0.999961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01033123727741079,"score_gpt":0.2893692455782085,"score_spread":0.2790380083007977,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}