{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:06:58Z","timestamp":1784300818528,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3803437.3806718","type":"proceedings-article","created":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:27:39Z","timestamp":1784298459000},"page":"1713-1720","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LLMSafeGuard: A Training-Free Framework for Safeguarding LLM Decoding via Context-Wise Similarity Validation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7569-0650","authenticated-orcid":false,"given":"Ximing","family":"Dong","sequence":"first","affiliation":[{"name":"Centre for Software Excellence, Huawei Canada, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3823-1771","authenticated-orcid":false,"given":"Shaowei","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Manitoba, Winnipeg, Manitoba, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4034-6650","authenticated-orcid":false,"given":"Dayi","family":"Lin","sequence":"additional","affiliation":[{"name":"Centre for Software Excellence, Huawei Canada, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7749-5513","authenticated-orcid":false,"given":"Ahmed E.","family":"Hassan","sequence":"additional","affiliation":[{"name":"Queen's University, Kingston, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n. d.]. https:\/\/www.perspectiveapi.com\/. Accessed: 2024-02-05."},{"key":"e_1_3_2_1_2_1","unstructured":"[n. d.]. https:\/\/huggingface.co\/docs\/transformers\/en\/perplexity. Accessed: 2024-02-05."},{"key":"e_1_3_2_1_3_1","unstructured":"2025. Code of LLMSafeGuard. https:\/\/anonymous.4open.science\/r\/realsafeguard-435C"},{"key":"e_1_3_2_1_4_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_5_1","unstructured":"Meta AI. [n. d.]. LLaMA-2 13B Chat. https:\/\/huggingface.co\/meta-llama\/Llama-2-13b-chat-hf. Accessed: 2025-08-02."},{"key":"e_1_3_2_1_6_1","unstructured":"Gabriel Alon and Michael Kamfonas. 2023. Detecting Language Model Attacks with Perplexity. arXiv:2308.14132 [cs.CL] https:\/\/arxiv.org\/abs\/2308.14132"},{"key":"e_1_3_2_1_7_1","unstructured":"Rohan Anil Andrew M Dai Orhan Firat Melvin Johnson Dmitry Lepikhin Alexandre Passos Siamak Shakeri Emanuel Taropa Paige Bailey Zhifeng Chen et al. 2023. Palm 2 technical report. arXiv preprint arXiv:2305.10403 (2023)."},{"key":"e_1_3_2_1_8_1","volume-title":"Jailbreaking black box large language models in twenty queries","author":"Chao Patrick","year":"2024","unstructured":"Patrick Chao, Alexander Robey, Edgar Dobriban, Hamed Hassani, George J Pappas, and Eric Wong. 2024. Jailbreaking black box large language models in twenty queries, 2024. URL https:\/\/arxiv.org\/abs\/2310.08419 1, 2 (2024), 3."},{"key":"e_1_3_2_1_9_1","unstructured":"Stanley F Chen Douglas Beeferman and Roni Rosenfeld. 1998. Evaluation metrics for language models. (1998)."},{"key":"e_1_3_2_1_10_1","volume-title":"Xing","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E. Gonzalez, Ion Stoica, and Eric P. Xing. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"e_1_3_2_1_11_1","volume-title":"nithum, and Will Cukierski","author":"Sorensen Jeffrey","year":"2017","unstructured":"cjadams, Jeffrey Sorensen, Julia Elliott, Lucas Dixon, Mark McDonald, nithum, and Will Cukierski. 2017. Toxic Comment Classification Challenge. Kaggle."},{"key":"e_1_3_2_1_12_1","volume-title":"Plug and play language models: A simple approach to controlled text generation. arXiv preprint arXiv:1912.02164","author":"Dathathri Sumanth","year":"2019","unstructured":"Sumanth Dathathri, Andrea Madotto, Janice Lan, Jane Hung, Eric Frank, Piero Molino, Jason Yosinski, and Rosanne Liu. 2019. Plug and play language models: A simple approach to controlled text generation. arXiv preprint arXiv:1912.02164 (2019)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00410"},{"key":"e_1_3_2_1_14_1","volume-title":"A survey of data augmentation approaches for NLP. arXiv preprint arXiv:2105.03075","author":"Feng Steven Y","year":"2021","unstructured":"Steven Y Feng, Varun Gangal, Jason Wei, Sarath Chandar, Soroush Vosoughi, Teruko Mitamura, and Eduard Hovy. 2021. A survey of data augmentation approaches for NLP. arXiv preprint arXiv:2105.03075 (2021)."},{"key":"e_1_3_2_1_15_1","unstructured":"Deep Ganguli Liane Lovitt Jackson Kernion Amanda Askell Yuntao Bai Saurav Kadavath Ben Mann Ethan Perez Nicholas Schiefer Kamal Ndousse et al. 2022. Red teaming language models to reduce harms: Methods scaling behaviors and lessons learned. arXiv preprint arXiv:2209.07858 (2022)."},{"key":"e_1_3_2_1_16_1","volume-title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations. arXiv:2312.06674 [cs.CL]","author":"Inan Hakan","year":"2023","unstructured":"Hakan Inan, Kartikeya Upasani, Jianfeng Chi, Rashi Rungta, Krithika Iyer, Yuning Mao, Michael Tontchev, Qing Hu, Brian Fuller, Davide Testuggine, and Madian Khabsa. 2023. Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations. arXiv:2312.06674 [cs.CL]"},{"key":"e_1_3_2_1_17_1","volume-title":"Preventing verbatim memorization in language models gives a false sense of privacy. arXiv preprint arXiv:2210.17546","author":"Ippolito Daphne","year":"2022","unstructured":"Daphne Ippolito, Florian Tram\u00e8r, Milad Nasr, Chiyuan Zhang, Matthew Jagielski, Katherine Lee, Christopher A Choquette-Choo, and Nicholas Carlini. 2022. Preventing verbatim memorization in language models gives a false sense of privacy. arXiv preprint arXiv:2210.17546 (2022)."},{"key":"e_1_3_2_1_18_1","volume-title":"Micah Goldblum, Aniruddha Saha, Jonas Geiping, and Tom Goldstein.","author":"Jain Neel","year":"2023","unstructured":"Neel Jain, Avi Schwarzschild, Yuxin Wen, Gowthami Somepalli, John Kirchenbauer, Ping yeh Chiang, Micah Goldblum, Aniruddha Saha, Jonas Geiping, and Tom Goldstein. 2023. Baseline Defenses for Adversarial Attacks Against Aligned Language Models. arXiv:2309.00614 [cs.LG] https:\/\/arxiv.org\/abs\/2309.00614"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.458"},{"key":"e_1_3_2_1_20_1","volume-title":"CTRL: A Conditional Transformer Language Model for Controllable Generation. arXiv:1909.05858 [cs.CL]","author":"Keskar Nitish Shirish","year":"2019","unstructured":"Nitish Shirish Keskar, Bryan McCann, Lav R. Varshney, Caiming Xiong, and Richard Socher. 2019. CTRL: A Conditional Transformer Language Model for Controllable Generation. arXiv:1909.05858 [cs.CL]"},{"key":"e_1_3_2_1_21_1","volume-title":"Critic-Guided Decoding for Controlled Text Generation. In Findings of the Association for Computational Linguistics: ACL","author":"Kim Minbeom","year":"2023","unstructured":"Minbeom Kim, Hwanhee Lee, Kang Min Yoo, Joonsuk Park, Hwaran Lee, and Kyomin Jung. 2023. Critic-Guided Decoding for Controlled Text Generation. In Findings of the Association for Computational Linguistics: ACL 2023. Association for Computational Linguistics, Toronto, Canada, 4598\u20134612."},{"key":"e_1_3_2_1_22_1","volume-title":"Bryan McCann, Nitish Shirish Keskar, Shafiq Joty, Richard Socher, and Nazneen Fatema Rajani.","author":"Krause Ben","year":"2021","unstructured":"Ben Krause, Akhilesh Deepak Gotmare, Bryan McCann, Nitish Shirish Keskar, Shafiq Joty, Richard Socher, and Nazneen Fatema Rajani. 2021. GeDi: Generative Discriminator Guided Sequence Generation. In Findings of the Association for Computational Linguistics: EMNLP 2021. Association for Computational Linguistics, Punta Cana, Dominican Republic, 4929\u20134952."},{"key":"e_1_3_2_1_23_1","volume-title":"Deepinception: Hypnotize large language model to be jailbreaker. arXiv preprint arXiv:2311.03191","author":"Li Xuan","year":"2023","unstructured":"Xuan Li, Zhanke Zhou, Jianing Zhu, Jiangchao Yao, Tongliang Liu, and Bo Han. 2023. Deepinception: Hypnotize large language model to be jailbreaker. arXiv preprint arXiv:2311.03191 (2023)."},{"key":"e_1_3_2_1_24_1","unstructured":"Percy Liang Rishi Bommasani Tony Lee Dimitris Tsipras Dilara Soylu Michihiro Yasunaga Yian Zhang Deepak Narayanan Yuhuai Wu Ananya Kumar et al. 2022. Holistic evaluation of language models. arXiv preprint arXiv:2211.09110 (2022)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.522"},{"key":"e_1_3_2_1_26_1","volume-title":"Autodan: Generating stealthy jailbreak prompts on aligned large language models. arXiv preprint arXiv:2310.04451","author":"Liu Xiaogeng","year":"2023","unstructured":"Xiaogeng Liu, Nan Xu, Muhao Chen, and Chaowei Xiao. 2023. Autodan: Generating stealthy jailbreak prompts on aligned large language models. arXiv preprint arXiv:2310.04451 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3605943"},{"key":"e_1_3_2_1_28_1","unstructured":"OpenAI. [n. d.]. GPT-2 Medium. https:\/\/huggingface.co\/openai-community\/gpt2-medium. Accessed: 2025-08-02."},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI. 2023. ChatGPT. https:\/\/chat.openai.com\/."},{"key":"e_1_3_2_1_30_1","volume-title":"Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693","author":"Qi Xiangyu","year":"2023","unstructured":"Xiangyu Qi, Yi Zeng, Tinghao Xie, Pin-Yu Chen, Ruoxi Jia, Prateek Mittal, and Peter Henderson. 2023. Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693 (2023)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.229"},{"key":"e_1_3_2_1_32_1","unstructured":"Nils Reimers and Iryna Gurevych. [n. d.]. Sentence Transformers: all-MiniLM-L6-v2. https:\/\/huggingface.co\/sentence-transformers\/all-MiniLM-L6-v2. Accessed: 2025-08-02."},{"key":"e_1_3_2_1_33_1","unstructured":"Llama Team. 2023. Llama-2-13b. https:\/\/huggingface.co\/meta-llama\/Llama-2-13b"},{"key":"e_1_3_2_1_34_1","unstructured":"Llama Team. 2023. Llama-2-7b-chat. https:\/\/huggingface.co\/meta-llama\/Llama-2-7b-chat-hf"},{"key":"e_1_3_2_1_35_1","unstructured":"Qdrant Team. [n. d.]. Qdrant Vector Database. https:\/\/qdrant.tech\/. Accessed: 2025-08-02."},{"key":"e_1_3_2_1_36_1","unstructured":"Qwen Team. 2024. Qwen2.5: A Party of Foundation Models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"e_1_3_2_1_37_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_38_1","volume-title":"Aakanksha Chowdhery, and Denny Zhou.","author":"Wang Xuezhi","year":"2022","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2022. Self-consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171 (2022)."},{"key":"e_1_3_2_1_39_1","volume-title":"Jailbreak and guard aligned language models with only few in-context demonstrations. arXiv preprint arXiv:2310.06387","author":"Wei Zeming","year":"2023","unstructured":"Zeming Wei, Yifei Wang, Ang Li, Yichuan Mo, and Yisen Wang. 2023. Jailbreak and guard aligned language models with only few in-context demonstrations. arXiv preprint arXiv:2310.06387 (2023)."},{"key":"e_1_3_2_1_40_1","unstructured":"Fangzhao Wu Yueqi Xie Jingwei Yi Jiawei Shao Justin Curl Lingjuan Lyu Qifeng Chen and Xing Xie. 2023. Defending ChatGPT against Jailbreak Attack via Self-Reminder. (2023)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Yueqi Xie Minghong Fang Renjie Pi and Neil Gong. 2024. GradSafe: Detecting Unsafe Prompts for LLMs via Safety-Critical Gradient Analysis. arXiv:2402.13494 [cs.CL]","DOI":"10.18653\/v1\/2024.acl-long.30"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-023-00765-8"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.276"},{"key":"e_1_3_2_1_44_1","volume-title":"GPT3Mix: Leveraging large-scale language models for text augmentation. arXiv preprint arXiv:2104.08826","author":"Yoo Kang Min","year":"2021","unstructured":"Kang Min Yoo, Dongju Park, Jaewook Kang, Sang-Woo Lee, and Woomyeong Park. 2021. GPT3Mix: Leveraging large-scale language models for text augmentation. arXiv preprint arXiv:2104.08826 (2021)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.773"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3617680"},{"key":"e_1_3_2_1_47_1","volume-title":"Fine-tuning language models from human preferences. arXiv preprint arXiv:1909.08593","author":"Ziegler Daniel M","year":"2019","unstructured":"Daniel M Ziegler, Nisan Stiennon, Jeffrey Wu, Tom B Brown, Alec Radford, Dario Amodei, Paul Christiano, and Geoffrey Irving. 2019. Fine-tuning language models from human preferences. arXiv preprint arXiv:1909.08593 (2019)."},{"key":"e_1_3_2_1_48_1","volume-title":"Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043","author":"Zou Andy","year":"2023","unstructured":"Andy Zou, Zifan Wang, Nicholas Carlini, Milad Nasr, J Zico Kolter, and Matt Fredrikson. 2023. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043 (2023)."}],"event":{"name":"FSE Companion '26: 34th ACM International Conference on the Foundations of Software Engineering","location":"Concordia University Montreal QC Canada","acronym":"FSE Companion '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 34th ACM International Conference on the Foundations of Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3803437.3806718","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:36:14Z","timestamp":1784298974000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3803437.3806718"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":48,"alternative-id":["10.1145\/3803437.3806718","10.1145\/3803437"],"URL":"https:\/\/doi.org\/10.1145\/3803437.3806718","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}