{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T12:09:47Z","timestamp":1784549387415,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","funder":[{"name":"the National Key Research and Development Program of China","award":["2023YFB3106504"],"award-info":[{"award-number":["2023YFB3106504"]}]},{"name":"Guangdong Provincial Key Laboratory of Novel Security Intelligence Technologies","award":["2022B1212010005"],"award-info":[{"award-number":["2022B1212010005"]}]},{"name":"the Major Key Project of PCL","award":["PCL2024A04"],"award-info":[{"award-number":["PCL2024A04"]}]},{"name":"Shenzhen Science and Technology Program","award":["ZDSYS20210623091809029, RCBS20221008093131089"],"award-info":[{"award-number":["ZDSYS20210623091809029, RCBS20221008093131089"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792438","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T12:38:33Z","timestamp":1777293513000},"page":"3078-3089","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["The Asymmetric Vulnerability: Bypassing LLM Defenses via Guardrail-Model Mismatch"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4131-6869","authenticated-orcid":false,"given":"Junyi","family":"Wang","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6289-8773","authenticated-orcid":false,"given":"Zhibin","family":"Zhu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0891-2544","authenticated-orcid":false,"given":"Chuanyi","family":"Liu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Josh Achiam et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774."},{"key":"e_1_3_2_1_2_1","unstructured":"Jianfeng Chi et al. 2024. Llama guard 3 vision: safeguarding human-ai image understanding conversations. arXiv preprint arXiv:2411.10414."},{"key":"e_1_3_2_1_3_1","volume-title":"Vaibhav Kumar, Faysal Hossain Shezan, Vinija Jain, and Aman Chadha.","author":"Chowdhury Arijit Ghosh","year":"2024","unstructured":"Arijit Ghosh Chowdhury, Md Mofijul Islam, Vaibhav Kumar, Faysal Hossain Shezan, Vinija Jain, and Aman Chadha. 2024. Breaking down the defenses: a comparative survey of attacks on large language models. arXiv preprint arXiv:2403.04786."},{"key":"e_1_3_2_1_4_1","unstructured":"Xun Deng Han Zhong Rui Ai Fuli Feng Zheng Wang and Xiangnan He. 2025. Less is more: improving llm alignment via preference data selection. arXiv preprint arXiv:2502.14560."},{"key":"e_1_3_2_1_5_1","unstructured":"Yi Dong Ronghui Mu Gaojie Jin Yi Qi Jinwei Hu Xingyu Zhao Jie Meng Wenjie Ruan and Xiaowei Huang. 2024. Building guardrails for large language models. arXiv preprint arXiv:2402.01822."},{"key":"e_1_3_2_1_6_1","unstructured":"Yi Dong et al. 2024. Safeguarding large language models: a survey. arXiv preprint arXiv:2406.02622."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2025.3582330"},{"key":"e_1_3_2_1_8_1","unstructured":"Abhimanyu Dubey et al. 2024. The llama 3 herd of models. arXiv e-prints arXiv--2407."},{"key":"e_1_3_2_1_9_1","unstructured":"William Hackett Lewis Birch Stefan Trawicki Neeraj Suri and Peter Garraghan. 2025. Bypassing prompt injection and jailbreak detection in llm guard-rails. arXiv preprint arXiv:2504.11168."},{"key":"e_1_3_2_1_10_1","unstructured":"Jiaqi Han Mingjian Jiang Yuxuan Song Stefano Ermon and Minkai Xu. 2024. f-po: generalizing preference optimization with f-divergence minimization. arXiv preprint arXiv:2410.21662."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0261"},{"key":"e_1_3_2_1_12_1","unstructured":"Binyuan Hui et al. 2024. Qwen2. 5-coder technical report. arXiv preprint arXiv:2409.12186."},{"key":"e_1_3_2_1_13_1","unstructured":"Hakan Inan et al. 2023. Llama guard: llm-based input-output safeguard for human-ai conversations. arXiv preprint arXiv:2312.06674."},{"key":"e_1_3_2_1_14_1","unstructured":"Jiaming Ji et al. 2023. Ai alignment: a comprehensive survey. arXiv preprint arXiv:2310.19852."},{"key":"e_1_3_2_1_15_1","unstructured":"Xiaojun Jia Tianyu Pang Chao Du Yihao Huang Jindong Gu Yang Liu Xiaochun Cao and Min Lin. 2024. Improved techniques for optimization-based jailbreaking on large language models. arXiv preprint arXiv:2405.21018."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671931"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650828"},{"key":"e_1_3_2_1_18_1","unstructured":"Martin Kuo Jianyi Zhang Aolin Ding Qinsi Wang Louis DiValentin Yujia Bao Wei Wei Hai Li and Yiran Chen. 2025. H-cot: hijacking the chain-of-thought safety reasoning mechanism to jailbreak large reasoning models including openai o1\/o3 deepseek-r1 and gemini 2.0 flash thinking. arXiv preprint arXiv:2502.12893."},{"key":"e_1_3_2_1_19_1","unstructured":"Jiacheng Liang Tanqiu Jiang Yuhui Wang Rongyi Zhu Fenglong Ma and Ting Wang. 2025. Autoran: weak-to-strong jailbreaking of large reasoning models. arXiv preprint arXiv:2505.10846."},{"key":"e_1_3_2_1_20_1","unstructured":"Aixin Liu et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437."},{"key":"e_1_3_2_1_21_1","unstructured":"Xiaogeng Liu Nan Xu Muhao Chen and Chaowei Xiao. 2023. Autodan: generating stealthy jailbreak prompts on aligned large language models. arXiv preprint arXiv:2310.04451."},{"key":"e_1_3_2_1_22_1","unstructured":"Yue Liu Xiaoxin He Miao Xiong Jinlan Fu Shumin Deng and Bryan Hooi. 2024. Flipattack: jailbreak llms via flipping. arXiv preprint arXiv:2410.02832."},{"key":"e_1_3_2_1_23_1","unstructured":"Yue Liu et al. 2025. Guardreasoner: towards reasoning-based llm safeguards. arXiv preprint arXiv:2501.18492."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714816"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26752"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1108\/IJPCC-07-2024-0224"},{"key":"e_1_3_2_1_27_1","unstructured":"Zhenxing Niu Haodong Ren Xinbo Gao Gang Hua and Rong Jin. 2024. Jailbreaking attack against multimodal large language model. arXiv preprint arXiv:2402.02309."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 27730--27744.","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_29_1","unstructured":"ShengYun Peng Pin-Yu Chen Jianfeng Chi Seongmin Lee and Duen Horng Chau. 2025. Shape it up! restoring llm safety during finetuning. arXiv preprint arXiv:2505.17196."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Rafael Rafailov Archit Sharma Eric Mitchell Christopher D Manning Stefano Ermon and Chelsea Finn. 2023. Direct preference optimization: your language model is secretly a reward model. Advances in neural information processing systems 36 53728--53741.","DOI":"10.52202\/075280-2338"},{"key":"e_1_3_2_1_31_1","unstructured":"Alexander Robey Eric Wong Hamed Hassani and George J Pappas. 2023. Smoothllm: defending large language models against jailbreaking attacks. arXiv preprint arXiv:2310.03684."},{"key":"e_1_3_2_1_32_1","unstructured":"Ephraiem Sarabamoun. 2025. Special-character adversarial attacks on open-source language model. arXiv preprint arXiv:2508.14070."},{"key":"e_1_3_2_1_33_1","unstructured":"Kasimir Schulz Kenneth Yeung and Kieran Evans. 2025. Tokenbreak: bypassing text classification models through token manipulation. arXiv preprint arXiv:2506.07948."},{"key":"e_1_3_2_1_34_1","unstructured":"Jintian Shao and Yiming Cheng. 2025. Cot is not true reasoning it is just a tight constraint to imitate: a theory perspective. arXiv preprint arXiv:2506.02878."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3670388"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3748239.3748242"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Xunguang Wang Zhenlan Ji Wenxuan Wang Zongjie Li Daoyuan Wu and Shuai Wang. 2025. Sok: evaluating jailbreak guardrails for large language models. arXiv preprint arXiv:2506.10597.","DOI":"10.1109\/SP63933.2026.00076"},{"key":"e_1_3_2_1_38_1","volume-title":"34th USENIX Security Symposium (USENIX Security 25)","author":"Xunguang","unstructured":"Xunguang Wang et al. 2025. {Selfdefend}:{llms} can defend themselves against jailbreaking in a practical manner. In 34th USENIX Security Symposium (USENIX Security 25), 2441--2460."},{"key":"e_1_3_2_1_39_1","volume-title":"Kiran Ramnath, Sougata Chaudhuri, Shubham Mehrotra, Xiang-Bo Mao, Sitaram Asur, et al.","author":"Wang Zhichao","year":"2024","unstructured":"Zhichao Wang, Bin Bi, Shiva Kumar Pentyala, Kiran Ramnath, Sougata Chaudhuri, Shubham Mehrotra, Xiang-Bo Mao, Sitaram Asur, et al. 2024. A comprehensive survey of llm alignment techniques: rlhf, rlaif, ppo, dpo and more. arXiv preprint arXiv:2407.16216."},{"key":"e_1_3_2_1_40_1","unstructured":"Zhipeng Wei Yuqi Liu and N Benjamin Erichson. 2024. Emoji attack: enhancing jailbreak attacks against judge llm detection. arXiv preprint arXiv:2411.01077."},{"key":"e_1_3_2_1_41_1","unstructured":"Jeff Wu Long Ouyang Daniel M Ziegler Nisan Stiennon Ryan Lowe Jan Leike and Paul Christiano. 2021. Recursively summarizing books with human feedback. arXiv preprint arXiv:2109.10862."},{"key":"e_1_3_2_1_42_1","volume-title":"ICLR 2025 Workshop on Foundation Models in the Wild.","author":"Xiang Zhen","unstructured":"Zhen Xiang, Shuang Yang, Nathaniel D Bastian, and Bo Li. [n.d.] Knowguard: robust reasoning enabled llm guardrail via knowledge-enhanced logical reasoning. In ICLR 2025 Workshop on Foundation Models in the Wild."},{"key":"e_1_3_2_1_43_1","unstructured":"Wenrui Xu and Keshab K Parhi. 2025. A survey of attacks on large language models. arXiv preprint arXiv:2505.12567."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714654"},{"key":"e_1_3_2_1_45_1","unstructured":"Zhongheng Yang Aijia Sun Yushang Zhao Yinuo Yang Dannier Li and Chengrui Zhou. 2025. Rlhf fine-tuning of llms for alignment with implicit user feedback in conversational recommenders. arXiv preprint arXiv:2508.05289."},{"key":"e_1_3_2_1_46_1","unstructured":"Sibo Yi Yule Liu Zhen Sun Tianshuo Cong Xinlei He Jiaxing Song Ke Xu and Qi Li. 2024. Jailbreak attacks and defenses against large language models: a survey. arXiv preprint arXiv:2407.04295."},{"key":"e_1_3_2_1_47_1","first-page":"137010","article-title":"Fincon: a synthesized llm multi-agent system with conceptual verbal reinforcement for enhanced financial decision making","volume":"37","author":"Yangyang Yu","year":"2024","unstructured":"Yangyang Yu et al. 2024. Fincon: a synthesized llm multi-agent system with conceptual verbal reinforcement for enhanced financial decision making. Advances in Neural Information Processing Systems, 37, 137010--137045.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_48_1","unstructured":"Zhuowen Yuan Zidi Xiong Yi Zeng Ning Yu Ruoxi Jia Dawn Song and Bo Li. 2024. Rigorllm: resilient guardrails for large language models against undesired content. arXiv preprint arXiv:2403.13031."},{"key":"e_1_3_2_1_49_1","unstructured":"Wenjun Zeng et al. 2024. Shieldgemma: generative ai content moderation based on gemma. arXiv preprint arXiv:2407.21772."},{"key":"e_1_3_2_1_50_1","unstructured":"Jingnan Zheng Xiangtian Ji Yijun Lu Chenhang Cui Weixiang Zhao Gelei Deng Zhenkai Liang An Zhang and Tat-Seng Chua. 2025. Rsafe: incentivizing proactive reasoning to build robust and adaptive llm safeguards. arXiv preprint arXiv:2506.07736."},{"key":"e_1_3_2_1_51_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J Zico Kolter and Matt Fredrikson. 2023. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792438","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:42:38Z","timestamp":1783150958000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792438"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":51,"alternative-id":["10.1145\/3774904.3792438","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792438","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}