{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T23:27:54Z","timestamp":1783553274871,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T00:00:00Z","timestamp":1763769600000},"content-version":"vor","delay-in-days":3,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Meta-BAIR Commons","award":["2024-2026"],"award-info":[{"award-number":["2024-2026"]}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2229876"],"award-info":[{"award-number":["2229876"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014895","name":"Open Philanthropy Project","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014895","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the Department of Homeland Security"},{"name":"IBM"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,19]]},"DOI":"10.1145\/3719027.3744836","type":"proceedings-article","created":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T23:32:38Z","timestamp":1763854358000},"page":"2833-2847","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["SecAlign: Defending Against Prompt Injection with Preference Optimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3274-6926","authenticated-orcid":false,"given":"Sizhe","family":"Chen","sequence":"first","affiliation":[{"name":"UC Berkeley, Berkeley, CA, USA and Meta, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2182-8291","authenticated-orcid":false,"given":"Arman","family":"Zharmagambetov","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6586-8378","authenticated-orcid":false,"given":"Saeed","family":"Mahloujifar","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9646-7710","authenticated-orcid":false,"given":"Kamalika","family":"Chaudhuri","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9944-9232","authenticated-orcid":false,"given":"David","family":"Wagner","sequence":"additional","affiliation":[{"name":"UC Berkeley, Berkeley, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1250-7306","authenticated-orcid":false,"given":"Chuan","family":"Guo","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"The llama 3 herd of models. arXiv:2407.21783","author":"Dubey Abhimanyu","year":"2024","unstructured":"Abhimanyu Dubey, Abhinav Jauhri, Abhinav Pandey, Abhishek Kadian, Ahmad Al-Dahle, Aiesha Letman, Akhil Mathur, Alan Schelten, Amy Yang, Angela Fan, et al. The llama 3 herd of models. arXiv:2407.21783, 2024."},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Wei Zeming","year":"2024","unstructured":"Zeming Wei, Yifei Wang, and Yisen Wang. Jailbreak and guard aligned language models with only few in-context demonstrations. In International Conference on Machine Learning (ICML), 2024."},{"key":"e_1_3_2_1_3_1","volume-title":"USENIX Security Symposium","author":"Chen Sizhe","year":"2025","unstructured":"Sizhe Chen, Julien Piet, Chawin Sitawarin, and David Wagner. Struq: Defending against prompt injection with structured queries. In USENIX Security Symposium, 2025."},{"key":"e_1_3_2_1_4_1","unstructured":"OpenAI. GPT-4 Technical Report 2023."},{"key":"e_1_3_2_1_5_1","first-page":"2","author":"Anthropic","year":"2023","unstructured":"Anthropic. Claude 2, 2023. URL https:\/\/www.anthropic.com\/index\/claude-2.","journal-title":"Claude"},{"key":"e_1_3_2_1_6_1","volume-title":"Llama 2: Open foundation and fine-tuned chat models. arXiv:2307.09288","author":"Hugo Touvron","year":"2023","unstructured":"Hugo Touvron et al. Llama 2: Open foundation and fine-tuned chat models. arXiv:2307.09288, 2023."},{"key":"e_1_3_2_1_7_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Debenedetti Edoardo","year":"2024","unstructured":"Edoardo Debenedetti, Jie Zhang, Mislav Balunovic, Luca Beurer-Kellner, Marc Fischer, and Florian Tram\u00e8r. Agentdojo: A dynamic environment to evaluate attacks and defenses for llm agents. In Advances in Neural Information Processing Systems (NeurIPS), 2024."},{"key":"e_1_3_2_1_8_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Drouin Alexandre","year":"2024","unstructured":"Alexandre Drouin, Maxime Gasse, Massimo Caccia, Issam H Laradji, Manuel Del Verme, Tom Marty, David Vazquez, Nicolas Chapados, and Alexandre Lacoste. Workarena: How capable are web agents at solving common knowledge work tasks? In International Conference on Machine Learning (ICML), 2024."},{"key":"e_1_3_2_1_9_1","volume-title":"Introducing computer use, a new claude 3.5 sonnet, and claude 3.5 haiku","year":"2024","unstructured":"Anthropic. Introducing computer use, a new claude 3.5 sonnet, and claude 3.5 haiku, 2024. URL https:\/\/www.anthropic.com\/news\/3--5-models-and-computer-use."},{"key":"e_1_3_2_1_10_1","volume-title":"Not what you've signed up for: Compromising real-world LLM-integrated applications with indirect prompt injection. arXiv:2302.12173","author":"Greshake Kai","year":"2023","unstructured":"Kai Greshake, Sahar Abdelnabi, Shailesh Mishra, Christoph Endres, Thorsten Holz, and Mario Fritz. Not what you've signed up for: Compromising real-world LLM-integrated applications with indirect prompt injection. arXiv:2302.12173, 2023."},{"key":"e_1_3_2_1_11_1","volume-title":"USENIX Security Symposium","author":"Liu Yupei","year":"2024","unstructured":"Yupei Liu, Yuqi Jia, Runpeng Geng, Jinyuan Jia, and Neil Zhenqiang Gong. Formalizing and benchmarking prompt injection attacks and defenses. In USENIX Security Symposium, 2024."},{"key":"e_1_3_2_1_12_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Toyer Sam","year":"2024","unstructured":"Sam Toyer, Olivia Watkins, Ethan Adrian Mendes, Justin Svegliato, Luke Bailey, Tiffany Wang, Isaac Ong, Karim Elmaaroufi, Pieter Abbeel, Trevor Darrell, Alan Ritter, and Stuart Russell. Tensor Trust: Interpretable Prompt Injection Attacks from an Online Game. In International Conference on Learning Representations (ICLR), 2024."},{"key":"e_1_3_2_1_13_1","volume-title":"Why openai is taking so long to launch agents. The Information","author":"Palazzolo Stephanie","year":"2025","unstructured":"Stephanie Palazzolo. Why openai is taking so long to launch agents. The Information, 2025. URL https:\/\/www.theinformation.com\/articles\/why-openai-is-taking-so-long-to-launch-agents."},{"key":"e_1_3_2_1_14_1","volume-title":"OWASP Top 10 for LLM Applications","author":"OWASP.","year":"2023","unstructured":"OWASP. OWASP Top 10 for LLM Applications, 2023. URL https:\/\/llmtop10.com."},{"key":"e_1_3_2_1_15_1","volume-title":"https:\/\/learnprompting.org","author":"Learn","year":"2023","unstructured":"Learn prompting. https:\/\/learnprompting.org, 2023."},{"key":"e_1_3_2_1_16_1","volume-title":"Delimiters won't save you from prompt injection","author":"Willison Simon","year":"2023","unstructured":"Simon Willison. Delimiters won't save you from prompt injection, 2023. URL https:\/\/simonwillison.net\/2023\/May\/11\/delimiters-wont-save-you."},{"key":"e_1_3_2_1_17_1","volume-title":"Benchmarking and defending against indirect prompt injection attacks on large language models. arXiv:2312.14197","author":"Yi Jingwei","year":"2023","unstructured":"Jingwei Yi, Yueqi Xie, Bin Zhu, Keegan Hines, Emre Kiciman, Guangzhong Sun, Xing Xie, and Fangzhao Wu. Benchmarking and defending against indirect prompt injection attacks on large language models. arXiv:2312.14197, 2023."},{"key":"e_1_3_2_1_18_1","volume-title":"European Symposium on Research in Computer Security (ESORICS)","author":"Piet Julien","year":"2023","unstructured":"Julien Piet, Maha Alrashed, Chawin Sitawarin, Sizhe Chen, Zeming Wei, Elizabeth Sun, Basel Alomair, and David Wagner. Jatmo: Prompt injection defense by task-specific finetuning. In European Symposium on Research in Computer Security (ESORICS), 2023."},{"key":"e_1_3_2_1_19_1","volume-title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions. arXiv:2404.13208","author":"Wallace Eric","year":"2024","unstructured":"Eric Wallace, Kai Xiao, Reimar Leike, Lilian Weng, Johannes Heidecke, and Alex Beutel. The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions. arXiv:2404.13208, 2024."},{"key":"e_1_3_2_1_20_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Wu Tong","year":"2025","unstructured":"Tong Wu, Shujian Zhang, Kaiqiang Song, Silei Xu, Sanqiang Zhao, Ravi Agrawal, Sathish Reddy Indurthi, Chong Xiang, Prateek Mittal, and Wenxuan Zhou. Instructional segment embedding: Improving llm safety with instruction hierarchy. In International Conference on Learning Representations (ICLR), 2025."},{"key":"e_1_3_2_1_21_1","volume-title":"Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043","author":"Zou Andy","year":"2023","unstructured":"Andy Zou, Zifan Wang, Nicholas Carlini, Milad Nasr, J Zico Kolter, and Matt Fredrikson. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043, 2023."},{"key":"e_1_3_2_1_22_1","volume-title":"Advprompter: Fast adaptive adversarial prompting for llms. arXiv:2404.16873","author":"Paulus Anselm","year":"2024","unstructured":"Anselm Paulus, Arman Zharmagambetov, Chuan Guo, Brandon Amos, and Yuandong Tian. Advprompter: Fast adaptive adversarial prompting for llms. arXiv:2404.16873, 2024."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689932.3694764"},{"key":"e_1_3_2_1_24_1","volume-title":"Meta SecAlign: A Secure Foundation LLM Against Prompt Injection Attacks. arXiv preprint arXiv:2507.02735","author":"Chen Sizhe","year":"2025","unstructured":"Sizhe Chen, Arman Zharmagambetov, David Wagner, and Chuan Guo. Meta SecAlign: A Secure Foundation LLM Against Prompt Injection Attacks. arXiv preprint arXiv:2507.02735, 2025."},{"key":"e_1_3_2_1_25_1","volume-title":"Lessons from defending gemini against indirect prompt injections","author":"Shi Chongyang","year":"2025","unstructured":"Chongyang Shi, Sharon Lin, Shuang Song, Jamie Hayes, Ilia Shumailov, Itay Yona, Juliette Pluto, Aneesh Pappu, Christopher A Choquette-Choo, Milad Nasr, et al. Lessons from defending gemini against indirect prompt injections. 2025."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.624"},{"key":"e_1_3_2_1_27_1","volume-title":"Prompt guard. https:\/\/llama.meta.com\/docs\/model-cards-and-prompt-formats\/prompt-guard","year":"2024","unstructured":"Meta. Prompt guard. https:\/\/llama.meta.com\/docs\/model-cards-and-prompt-formats\/prompt-guard, 2024."},{"key":"e_1_3_2_1_28_1","volume-title":"How Robust is Google's Bard to Adversarial Image Attacks? arXiv:2309.11751","author":"Dong Yinpeng","year":"2023","unstructured":"Yinpeng Dong, Huanran Chen, Jiawei Chen, Zhengwei Fang, Xiao Yang, Yichi Zhang, Yu Tian, Hang Su, and Jun Zhu. How Robust is Google's Bard to Adversarial Image Attacks? arXiv:2309.11751, 2023."},{"key":"e_1_3_2_1_29_1","volume-title":"Data exfiltration from slack ai via indirect prompt injection","year":"2024","unstructured":"PromptArmor. Data exfiltration from slack ai via indirect prompt injection, 2024. URL https:\/\/promptarmor.substack.com\/p\/data-exfiltration-from-slack-ai-via."},{"key":"e_1_3_2_1_30_1","volume-title":"https:\/\/slack.com","year":"2013","unstructured":"Salesforce. Slack. https:\/\/slack.com, 2013."},{"key":"e_1_3_2_1_31_1","volume-title":"https:\/\/embracethered.com\/blog\/posts\/2023\/google-bard-data-exfiltration","author":"Hacking","year":"2023","unstructured":"Hacking google bard - from prompt injection to data exfiltration. https:\/\/embracethered.com\/blog\/posts\/2023\/google-bard-data-exfiltration, 2023."},{"key":"e_1_3_2_1_32_1","volume-title":"From prompt injection to c2 with claude computer use. https:\/\/embracethered.com\/blog\/posts\/2024\/claude-computer-use-c2-the-zombais-are-coming","author":"Zombais","year":"2024","unstructured":"Zombais: From prompt injection to c2 with claude computer use. https:\/\/embracethered.com\/blog\/posts\/2024\/claude-computer-use-c2-the-zombais-are-coming, 2024."},{"key":"e_1_3_2_1_33_1","volume-title":"https:\/\/thehackernews.com\/2024\/09\/chatgpt-macos-flaw-couldve-enabled-long.html","author":"Chatgpt","year":"2024","unstructured":"Chatgpt macos flaw could've enabled long-term spyware via memory function. https:\/\/thehackernews.com\/2024\/09\/chatgpt-macos-flaw-couldve-enabled-long.html, 2024."},{"key":"e_1_3_2_1_34_1","volume-title":"Signed-prompt: A new approach to prevent prompt injection attacks against llm-integrated applications. arXiv:2401.07612","author":"Suo Xuchen","year":"2024","unstructured":"Xuchen Suo. Signed-prompt: A new approach to prevent prompt injection attacks against llm-integrated applications. arXiv:2401.07612, 2024."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.4236\/jsea.2024.171003"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CSDE59766.2023.10487667"},{"key":"e_1_3_2_1_37_1","volume-title":"ICML Workshop on Reliable and Responsible Foundation Models","author":"Chen Sizhe","year":"2025","unstructured":"Sizhe Chen, Yizhu Wang, Nicholas Carlini, Chawin Sitawarin, and David Wagner. Defending Against Prompt Injection with a Few DefensiveTokens. ICML Workshop on Reliable and Responsible Foundation Models, 2025."},{"key":"e_1_3_2_1_38_1","volume-title":"NeurIPS ML Safety Workshop","author":"Perez F\u00e1bio","year":"2022","unstructured":"F\u00e1bio Perez and Ian Ribeiro. Ignore previous prompt: Attack techniques for language models. In NeurIPS ML Safety Workshop, 2022."},{"key":"e_1_3_2_1_39_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Madry Aleksander","year":"2018","unstructured":"Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, and Adrian Vladu. Towards deep learning models resistant to adversarial attacks. In International Conference on Learning Representations (ICLR), 2018."},{"key":"e_1_3_2_1_40_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. Direct preference optimization: Your language model is secretly a reward model. In Advances in Neural Information Processing Systems (NeurIPS), 2024."},{"key":"e_1_3_2_1_41_1","first-page":"27730","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al. Training language models to follow instructions with human feedback. In Advances in Neural Information Processing Systems (NeurIPS), pages 27730--27744, 2022."},{"key":"e_1_3_2_1_42_1","volume-title":"KTO: Model alignment as prospect theoretic optimization. arXiv:2402.01306","author":"Ethayarajh Kawin","year":"2024","unstructured":"Kawin Ethayarajh, Winnie Xu, Niklas Muennighoff, Dan Jurafsky, and Douwe Kiela. KTO: Model alignment as prospect theoretic optimization. arXiv:2402.01306, 2024."},{"key":"e_1_3_2_1_43_1","volume-title":"ORPO: Monolithic Preference Optimization without Reference Model. arXiv:2403.07691","author":"Hong Jiwoo","year":"2024","unstructured":"Jiwoo Hong, Noah Lee, and James Thorne. ORPO: Monolithic Preference Optimization without Reference Model. arXiv:2403.07691, 2024."},{"key":"e_1_3_2_1_44_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Dubois Yann","year":"2024","unstructured":"Yann Dubois, Chen Xuechen Li, Rohan Taori, Tianyi Zhang, Ishaan Gulrajani, Jimmy Ba, Carlos Guestrin, Percy S Liang, and Tatsunori B Hashimoto. Alpacafarm: A simulation framework for methods that learn from human feedback. In Advances in Neural Information Processing Systems (NeurIPS), 2024."},{"key":"e_1_3_2_1_45_1","volume-title":"February","author":"Ruebsamen Gene","year":"2024","unstructured":"Gene Ruebsamen. Cleaned Alpaca Dataset, February 2024. URL https:\/\/github.com\/gururise\/AlpacaDataCleaned."},{"key":"e_1_3_2_1_46_1","volume-title":"AlpacaEval: An Automatic Evaluator of Instruction-following Models. https:\/\/github.com\/tatsu-lab\/alpaca_eval","author":"Li Xuechen","year":"2023","unstructured":"Xuechen Li, Tianyi Zhang, Yann Dubois, Rohan Taori, Ishaan Gulrajani, Carlos Guestrin, Percy Liang, and Tatsunori B. Hashimoto. AlpacaEval: An Automatic Evaluator of Instruction-following Models. https:\/\/github.com\/tatsu-lab\/alpaca_eval, 2023."},{"key":"e_1_3_2_1_47_1","volume-title":"https:\/\/github.com\/huggingface","author":"Hugging Face Inc. Huggingface.","year":"2021","unstructured":"Hugging Face Inc. Huggingface. https:\/\/github.com\/huggingface, 2021."},{"key":"e_1_3_2_1_48_1","first-page":"7B","author":"Jiang Albert Q.","year":"2023","unstructured":"Albert Q. Jiang et al. Mistral 7B, 2023. arXiv:2310.06825.","journal-title":"Mistral"},{"key":"e_1_3_2_1_49_1","volume-title":"LLaMA: Open and Efficient Foundation Language Models. arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, Aurelien Rodriguez, Armand Joulin, Edouard Grave, and Guillaume Lample. LLaMA: Open and Efficient Foundation Language Models. arXiv:2302.13971, 2023."},{"key":"e_1_3_2_1_50_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations (ICLR)","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et al. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations (ICLR), 2022."},{"key":"e_1_3_2_1_51_1","volume-title":"TRL: Transformer Reinforcement Learning. https:\/\/github.com\/huggingface\/trl","author":"von Werra Leandro","year":"2020","unstructured":"Leandro von Werra, Younes Belkada, Lewis Tunstall, Edward Beeching, Tristan Thrush, Nathan Lambert, and Shengyi Huang. TRL: Transformer Reinforcement Learning. https:\/\/github.com\/huggingface\/trl, 2020."},{"key":"e_1_3_2_1_52_1","volume-title":"PEFT: State-of-the-art Parameter-Efficient Fine-Tuning methods. https:\/\/github.com\/huggingface\/peft","author":"Mangrulkar Sourab","year":"2022","unstructured":"Sourab Mangrulkar, Sylvain Gugger, Lysandre Debut, Younes Belkada, Sayak Paul, and Benjamin Bossan. PEFT: State-of-the-art Parameter-Efficient Fine-Tuning methods. https:\/\/github.com\/huggingface\/peft, 2022."},{"key":"e_1_3_2_1_53_1","volume-title":"Pytorch FSDP: experiences on scaling fully sharded data parallel. arXiv:2304.11277","author":"Zhao Yanli","year":"2023","unstructured":"Yanli Zhao, Andrew Gu, Rohan Varma, Liang Luo, Chien-Chin Huang, Min Xu, Less Wright, Hamid Shojanazeri, Myle Ott, Sam Shleifer, et al. Pytorch FSDP: experiences on scaling fully sharded data parallel. arXiv:2304.11277, 2023."},{"key":"e_1_3_2_1_54_1","volume-title":"Gpt-4o mini: advancing cost-efficient intelligence. https:\/\/openai.com\/index\/gpt-4o-mini-advancing-cost-efficient-intelligence\/","author":"AI.","year":"2024","unstructured":"OpenAI. Gpt-4o mini: advancing cost-efficient intelligence. https:\/\/openai.com\/index\/gpt-4o-mini-advancing-cost-efficient-intelligence\/, 2024."},{"key":"e_1_3_2_1_55_1","volume-title":"Xing","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E. Gonzalez, Ion Stoica, and Eric P. Xing. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality, 2023."},{"key":"e_1_3_2_1_56_1","volume-title":"International Conference on Learning Representations (ICLR)","author":"Zverev Egor","year":"2025","unstructured":"Egor Zverev, Sahar Abdelnabi, Soroush Tabesh, Mario Fritz, and Christoph H Lampert. Can llms separate instructions from data? and what do we even mean by that? In International Conference on Learning Representations (ICLR), 2025."},{"key":"e_1_3_2_1_57_1","volume-title":"Many-shot jailbreaking. Advances in Neural Information Processing Systems (NeurIPS), 37: 129696--129742","author":"Anil Cem","year":"2024","unstructured":"Cem Anil, Esin Durmus, Nina Panickssery, Mrinank Sharma, Joe Benton, Sandipan Kundu, Joshua Batson, Meg Tong, Jesse Mu, Daniel Ford, et al. Many-shot jailbreaking. Advances in Neural Information Processing Systems (NeurIPS), 37: 129696--129742, 2024."},{"key":"e_1_3_2_1_58_1","volume-title":"Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300","author":"Hendrycks Dan","year":"2020","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300, 2020."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"},{"key":"e_1_3_2_1_60_1","volume-title":"Agieval: A human-centric benchmark for evaluating foundation models. arXiv preprint arXiv:2304.06364","author":"Zhong Wanjun","year":"2023","unstructured":"Wanjun Zhong, Ruixiang Cui, Yiduo Guo, Yaobo Liang, Shuai Lu, Yanlin Wang, Amin Saied, Weizhu Chen, and Nan Duan. Agieval: A human-centric benchmark for evaluating foundation models. arXiv preprint arXiv:2304.06364, 2023."},{"key":"e_1_3_2_1_61_1","volume-title":"Commonsenseqa: A question answering challenge targeting commonsense knowledge. arXiv preprint arXiv:1811.00937","author":"Talmor Alon","year":"2018","unstructured":"Alon Talmor, Jonathan Herzig, Nicholas Lourie, and Jonathan Berant. Commonsenseqa: A question answering challenge targeting commonsense knowledge. arXiv preprint arXiv:1811.00937, 2018."},{"key":"e_1_3_2_1_62_1","first-page":"24824","volume-title":"Denny Zhou, et al. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems (NeurIPS)","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems (NeurIPS), pages 24824--24837, 2022."},{"key":"e_1_3_2_1_63_1","volume-title":"Multilingual machine translation with large language models: Empirical results and analysis. arXiv:2304.04675","author":"Zhu Wenhao","year":"2023","unstructured":"Wenhao Zhu, Hongyi Liu, Qingxiu Dong, Jingjing Xu, Shujian Huang, Lingpeng Kong, Jiajun Chen, and Lei Li. Multilingual machine translation with large language models: Empirical results and analysis. arXiv:2304.04675, 2023."},{"key":"e_1_3_2_1_64_1","first-page":"39","volume-title":"Transactions of the Association for Computational Linguistics","author":"Zhang Tianyi","year":"2023","unstructured":"Tianyi Zhang, Faisal Ladhak, Esin Durmus, Percy Liang, Kathleen McKeown, and Tatsunori Hashimoto. Benchmarking large language models for news summarization. Transactions of the Association for Computational Linguistics, pages 39--57, 2023."},{"key":"e_1_3_2_1_65_1","volume-title":"The GPT store. https:\/\/chat.openai.com\/gpts","author":"AI.","year":"2024","unstructured":"OpenAI. The GPT store. https:\/\/chat.openai.com\/gpts, 2024."},{"key":"e_1_3_2_1_66_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","volume":"36","author":"Schick Timo","year":"2024","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess\u00ec, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. Toolformer: Language models can teach themselves to use tools. In Advances in Neural Information Processing Systems (NeurIPS), volume 36, 2024."},{"key":"e_1_3_2_1_67_1","volume-title":"Gorilla: Large language model connected with massive apis. arXiv:2305.15334","author":"Patil Shishir G","year":"2023","unstructured":"Shishir G Patil, Tianjun Zhang, Xin Wang, and Joseph E Gonzalez. Gorilla: Large language model connected with massive apis. arXiv:2305.15334, 2023."},{"key":"e_1_3_2_1_68_1","volume-title":"ChatGPT plugins. https:\/\/openai.com\/index\/chatgpt-plugins\/","author":"AI.","year":"2024","unstructured":"OpenAI. ChatGPT plugins. https:\/\/openai.com\/index\/chatgpt-plugins\/, 2024."},{"key":"e_1_3_2_1_69_1","unstructured":"Hezekiah J Branch Jonathan Rodriguez Cefalu Jeremy McHugh Leyla Hujer Aditya Bahl Daniel del Castillo Iglesias Ron Heichman and Ramesh Darwishi. Evaluating the susceptibility of pre-trained language models via handcrafted adversarial examples. arXiv:2209.02128 2022."},{"key":"e_1_3_2_1_70_1","first-page":"2311","volume":"200","author":"Yu Jiahao","year":"2023","unstructured":"Jiahao Yu, Yuhang Wu, Dong Shu, Mingyu Jin, and Xinyu Xing. Assessing Prompt Injection Risks in 200 Custom GPTs. arXiv:2311.11538, 2023.","journal-title":"Assessing Prompt Injection Risks in"},{"key":"e_1_3_2_1_71_1","volume-title":"Proceedings of the IEEE international symposium on secure software engineering","author":"Halfond William G","year":"2006","unstructured":"William G Halfond, Jeremy Viegas, Alessandro Orso, et al. A classification of SQL-injection attacks and countermeasures. In Proceedings of the IEEE international symposium on secure software engineering, 2006."},{"key":"e_1_3_2_1_72_1","volume-title":"Command injection | OWASP foundation","author":"Zhong Weilin","year":"2024","unstructured":"Weilin Zhong, Wichers, Amwestgate, Rezos, Clow808, KristenS, Jason Li, Andrew Smith, Jmanico, Tal Mel, and kingthorin. Command injection | OWASP foundation, 2024."},{"key":"e_1_3_2_1_73_1","volume-title":"International Conference on Machine Learning (ICML)","author":"Mazeika Mantas","year":"2024","unstructured":"Mantas Mazeika, Long Phan, Xuwang Yin, Andy Zou, Zifan Wang, Norman Mu, Elham Sakhaee, Nathaniel Li, Steven Basart, Bo Li, et al. Harmbench: A standardized evaluation framework for automated red teaming and robust refusal. In International Conference on Machine Learning (ICML), 2024."},{"key":"e_1_3_2_1_74_1","first-page":"2633","volume-title":"USENIX Security Symposium","author":"Carlini Nicholas","year":"2021","unstructured":"Nicholas Carlini, Florian Tramer, Eric Wallace, Matthew Jagielski, Ariel Herbert-Voss, Katherine Lee, Adam Roberts, Tom Brown, Dawn Song, Ulfar Erlingsson, et al. Extracting training data from large language models. In USENIX Security Symposium, pages 2633--2650, 2021."},{"key":"e_1_3_2_1_75_1","first-page":"40306","volume-title":"International Conference on Machine Learning (ICML)","author":"Yu Weichen","year":"2023","unstructured":"Weichen Yu, Tianyu Pang, Qian Liu, Chao Du, Bingyi Kang, Yan Huang, Min Lin, and Shuicheng Yan. Bag of tricks for training data extraction from language models. In International Conference on Machine Learning (ICML), pages 40306--40320, 2023."},{"key":"e_1_3_2_1_76_1","volume-title":"Scalable extraction of training data from (production) language models. arXiv:2311.17035","author":"Nasr Milad","year":"2023","unstructured":"Milad Nasr, Nicholas Carlini, Jonathan Hayase, Matthew Jagielski, A Feder Cooper, Daphne Ippolito, Christopher A Choquette-Choo, Eric Wallace, Florian Tram\u00e8r, and Katherine Lee. Scalable extraction of training data from (production) language models. arXiv:2311.17035, 2023."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP46215.2023.10179300"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.272"},{"key":"e_1_3_2_1_79_1","volume-title":"Membership inference attacks against language models via neighbourhood comparison. arXiv:2305.18462","author":"Mattern Justus","year":"2023","unstructured":"Justus Mattern, Fatemehsadat Mireshghallah, Zhijing Jin, Bernhard Sch\u00f6lkopf, Mrinmaya Sachan, and Taylor Berg-Kirkpatrick. Membership inference attacks against language models via neighbourhood comparison. arXiv:2305.18462, 2023."},{"key":"e_1_3_2_1_80_1","volume-title":"Do membership inference attacks work on large language models? arXiv:2402.07841","author":"Duan Michael","year":"2024","unstructured":"Michael Duan, Anshuman Suri, Niloofar Mireshghallah, Sewon Min, Weijia Shi, Luke Zettlemoyer, Yulia Tsvetkov, Yejin Choi, David Evans, and Hannaneh Hajishirzi. Do membership inference attacks work on large language models? arXiv:2402.07841, 2024."},{"key":"e_1_3_2_1_81_1","volume-title":"PromptBench: Towards Evaluating the Robustness of Large Language Models on Adversarial Prompts. arXiv:2306.04528","author":"Kaijie Zhu","year":"2023","unstructured":"Kaijie Zhu et al. PromptBench: Towards Evaluating the Robustness of Large Language Models on Adversarial Prompts. arXiv:2306.04528, 2023."},{"key":"e_1_3_2_1_82_1","volume-title":"Nicholas Carlini. Backdoor Attacks for In-Context Learning with Language Models. In ICML Workshop on Adversarial Machine Learning","author":"Kandpal Nikhil","year":"2023","unstructured":"Nikhil Kandpal, Matthew Jagielski, Florian Tram\u00e8r, and Nicholas Carlini. Backdoor Attacks for In-Context Learning with Language Models. In ICML Workshop on Adversarial Machine Learning, 2023."},{"key":"e_1_3_2_1_83_1","volume-title":"On the Robustness of ChatGPT: An Adversarial and Out-of-distribution Perspective. ICLR 2023 Workshop on Trustworthy and Reliable Large-Scale Machine Learning Models","author":"Jindong","year":"2023","unstructured":"Jindong Wang et al. On the Robustness of ChatGPT: An Adversarial and Out-of-distribution Perspective. ICLR 2023 Workshop on Trustworthy and Reliable Large-Scale Machine Learning Models, 2023."},{"key":"e_1_3_2_1_84_1","volume-title":"A survey of reinforcement learning from human feedback. arXiv:2312.14925","author":"Kaufmann Timo","year":"2023","unstructured":"Timo Kaufmann, Paul Weng, Viktor Bengs, and Eyke H\u00fcllermeier. A survey of reinforcement learning from human feedback. arXiv:2312.14925, 2023."},{"key":"e_1_3_2_1_85_1","first-page":"229","volume-title":"Machine Learning","author":"Williams Ronald J.","year":"1992","unstructured":"Ronald J. Williams. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine Learning, pages 229--256, 1992."},{"key":"e_1_3_2_1_86_1","volume-title":"Proximal policy optimization algorithms. arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. Proximal policy optimization algorithms. arXiv:1707.06347, 2017."},{"key":"e_1_3_2_1_87_1","volume-title":"RLHF workflow: From reward modeling to online RLHF. arXiv:2405.07863","author":"Dong Hanze","year":"2024","unstructured":"Hanze Dong, Wei Xiong, Bo Pang, Haoxiang Wang, Han Zhao, Yingbo Zhou, Nan Jiang, Doyen Sahoo, Caiming Xiong, and Tong Zhang. RLHF workflow: From reward modeling to online RLHF. arXiv:2405.07863, 2024."},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/SaTML64287.2025.00011"},{"key":"e_1_3_2_1_89_1","volume-title":"https:\/\/techcommunity.microsoft.com\/t5\/ai-azure-ai-services-blog\/azure-ai-announces-prompt-shields-for-jailbreak-and-indirect\/ba-p\/4099140","author":"Prompt","year":"2024","unstructured":"Prompt shields in azure ai. https:\/\/techcommunity.microsoft.com\/t5\/ai-azure-ai-services-blog\/azure-ai-announces-prompt-shields-for-jailbreak-and-indirect\/ba-p\/4099140, 2024."},{"key":"e_1_3_2_1_90_1","volume-title":"Baseline defenses for adversarial attacks against aligned language models. arXiv:2309.00614","author":"Jain Neel","year":"2023","unstructured":"Neel Jain, Avi Schwarzschild, Yuxin Wen, Gowthami Somepalli, John Kirchenbauer, Ping-yeh Chiang, Micah Goldblum, Aniruddha Saha, Jonas Geiping, and Tom Goldstein. Baseline defenses for adversarial attacks against aligned language models. arXiv:2309.00614, 2023."},{"key":"e_1_3_2_1_91_1","volume-title":"Effectively controlling reasoning models through thinking intervention. arXiv preprint arXiv:2503.24370","author":"Wu Tong","year":"2025","unstructured":"Tong Wu, Chong Xiang, Jiachen T Wang, and Prateek Mittal. Effectively controlling reasoning models through thinking intervention. arXiv preprint arXiv:2503.24370, 2025."},{"key":"e_1_3_2_1_92_1","volume-title":"Defeating prompt injections by design. arXiv preprint arXiv:2503.18813","author":"Debenedetti Edoardo","year":"2025","unstructured":"Edoardo Debenedetti, Ilia Shumailov, Tianqi Fan, Jamie Hayes, Nicholas Carlini, Daniel Fabian, Christoph Kern, Chongyang Shi, Andreas Terzis, and Florian Tram\u00e8r. Defeating prompt injections by design. arXiv preprint arXiv:2503.18813, 2025."},{"key":"e_1_3_2_1_93_1","volume-title":"Multi-modal prompt injection image attacks against GPT-4V","author":"Willison Simon","year":"2023","unstructured":"Simon Willison. Multi-modal prompt injection image attacks against GPT-4V, 2023. URL https:\/\/simonwillison.net\/2023\/Oct\/14\/multi-modal-prompt-injection."},{"key":"e_1_3_2_1_94_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Carlini Nicholas","year":"2024","unstructured":"Nicholas Carlini, Milad Nasr, Christopher A Choquette-Choo, Matthew Jagielski, Irena Gao, Pang Wei W Koh, Daphne Ippolito, Florian Tramer, and Ludwig Schmidt. Are aligned neural networks adversarially aligned? Advances in Neural Information Processing Systems (NeurIPS), 2024."}],"event":{"name":"CCS '25: ACM SIGSAC Conference on Computer and Communications Security","location":"Taipei Taiwan","acronym":"CCS '25","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2025 ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3719027.3744836","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3719027.3744836","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,22]],"date-time":"2025-12-22T22:12:20Z","timestamp":1766441540000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3719027.3744836"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,19]]},"references-count":94,"alternative-id":["10.1145\/3719027.3744836","10.1145\/3719027"],"URL":"https:\/\/doi.org\/10.1145\/3719027.3744836","relation":{},"subject":[],"published":{"date-parts":[[2025,11,19]]},"assertion":[{"value":"2025-11-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}