{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:18:13Z","timestamp":1783153093699,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62402183"],"award-info":[{"award-number":["62402183"]}]},{"name":"National Natural Science Foundation of China","award":["92582108"],"award-info":[{"award-number":["92582108"]}]},{"name":"Shanghai Special Program for Promoting High-Quality Industrial Development","award":["250668"],"award-info":[{"award-number":["250668"]}]},{"name":"Shanghai Special Program for Promoting High-Quality Industrial Development","award":["250203"],"award-info":[{"award-number":["250203"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792204","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T13:28:36Z","timestamp":1777296516000},"page":"2683-2694","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PADD: Prefix-based Attention Divergence Detector for LLM Jailbreaks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0409-2614","authenticated-orcid":false,"given":"Ziqun","family":"Bao","sequence":"first","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8144-3966","authenticated-orcid":false,"given":"Jiaqiang","family":"Niu","sequence":"additional","affiliation":[{"name":"Zhejiang Sci-Tech University, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0414-7521","authenticated-orcid":false,"given":"Yuchen","family":"Shao","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China and Shanghai Innovation Institute, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9162-9688","authenticated-orcid":false,"given":"Chengcheng","family":"Wan","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China, China and Shanghai Innovation Institute, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Detecting language model attacks with perplexity. arXiv:2308.14132","author":"Alon Gabriel","year":"2023","unstructured":"Gabriel Alon and Michael Kamfonas. 2023. Detecting language model attacks with perplexity. arXiv:2308.14132 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Ahsan Ayub and Subhabrata Majumdar","year":"2024","unstructured":"Md. Ahsan Ayub and Subhabrata Majumdar. 2024. Embedding-based classifiers can detect prompt injection attacks. https:\/\/arxiv.org\/abs\/2410.22284"},{"key":"e_1_3_2_1_3_1","volume-title":"Constitutional AI: Harmlessness from AI Feedback. https:\/\/arxiv.org\/abs\/2212.08073","author":"Bai Yuntao","year":"2022","unstructured":"Yuntao Bai, Saurav Kadavath, Sandipan Kundu, and Amanda Askell., 2022. Constitutional AI: Harmlessness from AI Feedback. https:\/\/arxiv.org\/abs\/2212.08073"},{"key":"e_1_3_2_1_4_1","volume-title":"Defending against alignment-breaking attacks via robustly aligned llm. arXiv:2309.14348","author":"Cao Bochuan","year":"2023","unstructured":"Bochuan Cao, Yuanpu Cao, Lu Lin, and Jinghui Chen. 2023. Defending against alignment-breaking attacks via robustly aligned llm. arXiv:2309.14348 (2023)."},{"key":"e_1_3_2_1_5_1","first-page":"55005","article-title":"Jailbreakbench: An open robustness benchmark for jailbreaking large language models","volume":"37","author":"Chao Patrick","year":"2024","unstructured":"Patrick Chao, Edoardo Debenedetti, Alexander Robey, Maksym Andriushchenko, Francesco Croce, Vikash Sehwag, Edgar Dobriban, Nicolas Flammarion, George J Pappas, Florian Tramer, et al., 2024. Jailbreakbench: An open robustness benchmark for jailbreaking large language models. NeurIPS, Vol. 37 (2024), 55005-55029.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_6_1","first-page":"23","article-title":"Jailbreaking black box large language models in twenty queries","author":"Chao Patrick","year":"2025","unstructured":"Patrick Chao, Alexander Robey, and \u03b7l. 2025. Jailbreaking black box large language models in twenty queries. In SaTML. IEEE, 23-42.","journal-title":"SaTML. IEEE"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.771"},{"key":"e_1_3_2_1_8_1","volume-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality.","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang et al., 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. (2023). https:\/\/arxiv.org\/abs\/2306.05685"},{"key":"e_1_3_2_1_9_1","volume-title":"Safe rlhf: Safe reinforcement learning from human feedback. arXiv:2310.12773","author":"Dai Josef","year":"2023","unstructured":"Josef Dai, Xuehai Pan, Ruiyang Sun, Jiaming Ji, Xinbo Xu, Mickel Liu, Yizhou Wang, and Yaodong Yang. 2023. Safe rlhf: Safe reinforcement learning from human feedback. arXiv:2310.12773 (2023)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2024.24188"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Y. Gong and colleagues. 2025. Safety Misalignment Against Large Language Models. In NDSS. https:\/\/www.ndss-symposium.org\/wp-content\/uploads\/2025-1089-paper.pdf","DOI":"10.14722\/ndss.2025.241089"},{"key":"e_1_3_2_1_12_1","volume-title":"Rohith Kuditipudi, Percy Liang, and Tatsunori Hashimoto.","author":"Gu Chenchen","year":"2025","unstructured":"Chenchen Gu, Xiang Lisa Li, Rohith Kuditipudi, Percy Liang, and Tatsunori Hashimoto. 2025. Auditing Prompt Caching in Language Model APIs. arXiv:2502.07776 (2025)."},{"key":"e_1_3_2_1_13_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi Wenfeng Wu Yuxiang Liu et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. (2025). https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4011"},{"key":"e_1_3_2_1_15_1","unstructured":"Zhengmian Hu Gang Wu Saayan Mitra Ruiyi Zhang and \u03b7l. 2024b. Token-Level Adversarial Prompt Detection Based on Perplexity Measures and Contextual Information. https:\/\/arxiv.org\/abs\/2311.11509"},{"key":"e_1_3_2_1_16_1","first-page":"341","article-title":"Promptshield: Deployable detection for prompt injection attacks","author":"Jacob Dennis","year":"2024","unstructured":"Dennis Jacob, Hend Alzahrani, Zhanhao Hu, Basel Alomair, and David Wagner. 2024. Promptshield: Deployable detection for prompt injection attacks. In CODASPY. 341-352.","journal-title":"CODASPY."},{"key":"e_1_3_2_1_17_1","volume-title":"Baseline defenses for adversarial attacks against aligned language models. arXiv:2309.00614","author":"Jain Neel","year":"2023","unstructured":"Neel Jain, Avi Schwarzschild, Yuxin Wen, Gowthami Somepalli, and Kirchenbauer., 2023. Baseline defenses for adversarial attacks against aligned language models. arXiv:2309.00614 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"Multitask Mayhem: Unveiling and Mitigating Safety Gaps in LLMs Fine-tuning. https:\/\/arxiv.org\/abs\/2409.15361","author":"Jan Essa","year":"2024","unstructured":"Essa Jan, Nouar AlDahoul, Moiz Ali, Faizan Ahmad, Fareed Zaffar, and Yasir Zaki. 2024. Multitask Mayhem: Unveiling and Mitigating Safety Gaps in LLMs Fine-tuning. https:\/\/arxiv.org\/abs\/2409.15361"},{"key":"e_1_3_2_1_19_1","first-page":"159","article-title":"Robust safety classifier against jailbreaking attacks: Adversarial prompt shield","author":"Kim Jinhwa","year":"2024","unstructured":"Jinhwa Kim, Ali Derakhshan, and Ian Harris. 2024. Robust safety classifier against jailbreaking attacks: Adversarial prompt shield. In WOAH. 159-170.","journal-title":"WOAH."},{"key":"e_1_3_2_1_20_1","volume-title":"Soheil Feizi, and Himabindu Lakkaraju.","author":"Kumar Aounon","year":"2025","unstructured":"Aounon Kumar, Chirag Agarwal, Suraj Srinivas, Aaron Jiaxun Li, Soheil Feizi, and Himabindu Lakkaraju. 2025. Certifying LLM Safety against Adversarial Prompting. https:\/\/arxiv.org\/abs\/2309.02705"},{"key":"e_1_3_2_1_21_1","volume-title":"Deepinception: Hypnotize large language model to be jailbreaker. arXiv:2311.03191","author":"Li Xuan","year":"2023","unstructured":"Xuan Li, Zhanke Zhou, Jianing Zhu, Jiangchao Yao, Tongliang Liu, and Bo Han. 2023. Deepinception: Hypnotize large language model to be jailbreaker. arXiv:2311.03191 (2023)."},{"key":"e_1_3_2_1_22_1","unstructured":"Xiaogeng Liu Nan Xu Muhao Chen and Chaowei Xiao. 2024. AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models. In ICLR. https:\/\/openreview.net\/forum?id=7Jwpw4qKkb"},{"key":"e_1_3_2_1_23_1","volume-title":"Jailbreaking chatgpt via prompt engineering: An empirical study. arXiv:2305.13860","author":"Liu Yi","year":"2023","unstructured":"Yi Liu, Gelei Deng, Zhengzi Xu, Yuekang Li, Yaowen Zheng, Ying Zhang, Lida Zhao, Tianwei Zhang, Kailong Wang, and Yang Liu. 2023. Jailbreaking chatgpt via prompt engineering: An empirical study. arXiv:2305.13860 (2023)."},{"key":"e_1_3_2_1_24_1","unstructured":"AI @ Meta Llama Team. 2024. The Llama 3 Herd of Models. https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_25_1","volume-title":"Jailbreakv: A benchmark for assessing the robustness of multimodal large language models against jailbreak attacks. arXiv:2404.03027","author":"Luo Weidi","year":"2024","unstructured":"Weidi Luo, Siyuan Ma, Xiaogeng Liu, Xiaoyu Guo, and Chaowei Xiao. 2024. Jailbreakv: A benchmark for assessing the robustness of multimodal large language models against jailbreak attacks. arXiv:2404.03027 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Shadow in the cache: Unveiling and mitigating privacy risks of kv-cache in llm inference. arXiv:2508.09442","author":"Luo Zhifan","year":"2025","unstructured":"Zhifan Luo, Shuo Shao, Su Zhang, Lijing Zhou, Yuke Hu, Chenxu Zhao, Zhihao Liu, and Zhan Qin. 2025. Shadow in the cache: Unveiling and mitigating privacy risks of kv-cache in llm inference. arXiv:2508.09442 (2025)."},{"key":"e_1_3_2_1_27_1","unstructured":"Long Ouyang Jeff Wu and \u03b7l. 2022. Training language models to follow instructions with human feedback. https:\/\/arxiv.org\/abs\/2203.02155"},{"key":"e_1_3_2_1_28_1","volume-title":"arXiv:2308.07308","author":"Phute Mansi","year":"2023","unstructured":"Mansi Phute, Alec Helbling, Matthew Hull, ShengYun Peng, Sebastian Szyller, Cory Cornelius, and Duen Horng Chau. 2023. LLM Self Defense: By Self Examination, LLMs Know They Are Being Tricked. arXiv:2308.07308 (2023)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3717659"},{"key":"e_1_3_2_1_30_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems, Vol. 36 (2023), 53728-53741."},{"key":"e_1_3_2_1_31_1","volume-title":"Smoothllm: Defending large language models against jailbreaking attacks. arXiv:2310.03684","author":"Robey Alexander","year":"2023","unstructured":"Alexander Robey, Eric Wong, Hamed Hassani, and George J Pappas. 2023. Smoothllm: Defending large language models against jailbreaking attacks. arXiv:2310.03684 (2023)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.301"},{"key":"e_1_3_2_1_33_1","unstructured":"Tiziano Santilli Marco De Luca Domenico Amalfitano Anna Rita Fasolino and Patrizio Pelliccione. [n.d.]. A Decontextualized LLM-based Safeguard Technique for Automated Jailbreak Mitigation. Available at SSRN 5503124 ([n.d.])."},{"key":"e_1_3_2_1_34_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv:2402.03300","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, and \u03b7l. 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_1_35_1","unstructured":"Mrinank Sharma Meg Tong Jesse Mu Jerry Wei Jorrit Kruthoff Scott Goodfriend Euan Ong Alwin Peng Raj Agarwal Cem Anil et al. 2025. Constitutional classifiers: Defending against universal jailbreaks across thousands of hours of red teaming. arXiv:2501.18837 (2025)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3670388"},{"key":"e_1_3_2_1_37_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard and \u03b7l. 2023. LLaMA: Open and Efficient Foundation Language Models. (2023). https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_38_1","volume-title":"Self-guard: Empower the llm to safeguard itself. arXiv:2310.15851","author":"Wang Zezhong","year":"2023","unstructured":"Zezhong Wang, Fangkai Yang, Lu Wang, Pu Zhao, Hongru Wang, Liang Chen, Qingwei Lin, and Kam-Fai Wong. 2023. Self-guard: Empower the llm to safeguard itself. arXiv:2310.15851 (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"Jailbroken: How Does LLM Safety Training Fail?. In NeurIPS. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/fd6613131889a4b656206c50a8bd7790-Paper-Conference.pdf","author":"Wei Alexander","year":"2023","unstructured":"Alexander Wei, Nika Haghtalab, and Jacob Steinhardt. 2023. Jailbroken: How Does LLM Safety Training Fail?. In NeurIPS. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/fd6613131889a4b656206c50a8bd7790-Paper-Conference.pdf"},{"key":"e_1_3_2_1_40_1","unstructured":"Daoyuan Wu Shuai Wang Yang Liu and Ning Liu. 2024. LLMs Can Defend Themselves Against Jailbreaking in a Practical Manner: A Vision Paper. https:\/\/arxiv.org\/abs\/2402.15727"},{"key":"e_1_3_2_1_41_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui and Zheng. 2025. Qwen3 Technical Report. (2025). https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_1_42_1","unstructured":"Zhiyuan Yu Xiaogeng Liu and \u03b7l. 2024. Don't Listen to Me: Understanding and Exploring Jailbreak Prompts of Large Language Models. In USENIX Security. https:\/\/www.usenix.org\/system\/files\/sec24fall-prepub-1500-yu-zhiyuan.pdf"},{"key":"e_1_3_2_1_43_1","volume-title":"Refuse whenever you feel unsafe: Improving safety in llms via decoupled refusal training. arXiv:2407.09121","author":"Yuan Youliang","year":"2024","unstructured":"Youliang Yuan, Wenxiang Jiao, Wenxuan Wang, Jen-tse Huang, Jiahao Xu, Tian Liang, Pinjia He, and Zhaopeng Tu. 2024a. Refuse whenever you feel unsafe: Improving safety in llms via decoupled refusal training. arXiv:2407.09121 (2024)."},{"key":"e_1_3_2_1_44_1","volume-title":"From hard refusals to safe-completions: Toward output-centric safety training. arXiv:2508.09224","author":"Yuan Yuan","year":"2025","unstructured":"Yuan Yuan, Tina Sriskandarajah, Anna-Luisa Brakman, Alec Helyar, Alex Beutel, Andrea Vallone, and Saachi Jain. 2025. From hard refusals to safe-completions: Toward output-centric safety training. arXiv:2508.09224 (2025)."},{"key":"e_1_3_2_1_45_1","volume-title":"Rigorllm: Resilient guardrails for large language models against undesired content. arXiv:2403.13031","author":"Yuan Zhuowen","year":"2024","unstructured":"Zhuowen Yuan, Zidi Xiong, and \u03b7l. 2024b. Rigorllm: Resilient guardrails for large language models against undesired content. arXiv:2403.13031 (2024)."},{"key":"e_1_3_2_1_46_1","unstructured":"Chongwen Zhao Zhihao Dou and Kaizhu Huang. 2025. Defending against Jailbreak through Early Exit Generation of Large Language Models. https:\/\/arxiv.org\/abs\/2408.11308"},{"key":"e_1_3_2_1_47_1","volume-title":"ICML (PMLR","volume":"61613","author":"Zheng Chujie","year":"2024","unstructured":"Chujie Zheng, Fan Yin, Hao Zhou, Fandong Meng, Jie Zhou, Kai-Wei Chang, Minlie Huang, and Nanyun Peng. 2024. On Prompt-Driven Safeguarding for Large Language Models. In ICML (PMLR, Vol. 235). 61593-61613. https:\/\/proceedings.mlr.press\/v235\/zheng24n.html"},{"key":"e_1_3_2_1_48_1","first-page":"40184","article-title":"Robust prompt optimization for defending language models against jailbreaking attacks","volume":"37","author":"Zhou Andy","year":"2024","unstructured":"Andy Zhou, Bo Li, and Haohan Wang. 2024a. Robust prompt optimization for defending language models against jailbreaking attacks. NeurIPS, Vol. 37 (2024), 40184-40211.","journal-title":"NeurIPS"},{"key":"e_1_3_2_1_49_1","volume-title":"Don't say no: Jailbreaking llm by suppressing refusal. arXiv:2404.16369","author":"Zhou Yukai","year":"2024","unstructured":"Yukai Zhou, Jian Lou, Zhijie Huang, Zhan Qin, and \u03b7l. 2024b. Don't say no: Jailbreaking llm by suppressing refusal. arXiv:2404.16369 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"GRAIT: Gradient-Driven Refusal-Aware Instruction Tuning for Effective Hallucination Mitigation. https:\/\/arxiv.org\/abs\/2502.05911","author":"Zhu Runchuan","year":"2025","unstructured":"Runchuan Zhu, Zinco Jiang, Jiang Wu, Zhipeng Ma, and \u03b7l. 2025. GRAIT: Gradient-Driven Refusal-Aware Instruction Tuning for Effective Hallucination Mitigation. https:\/\/arxiv.org\/abs\/2502.05911"},{"key":"e_1_3_2_1_51_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J. Zico Kolter and Matt Fredrikson. 2023. Universal and Transferable Adversarial Attacks on Aligned Language Models. (2023). https:\/\/arxiv.org\/abs\/2307.15043"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792204","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:42:37Z","timestamp":1783150957000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792204"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":51,"alternative-id":["10.1145\/3774904.3792204","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792204","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}