{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T02:34:17Z","timestamp":1784342057293,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":119,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3736561","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T20:52:41Z","timestamp":1754254361000},"page":"6216-6226","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":24,"title":["A Survey on Trustworthy LLM Agents: Threats and Countermeasures"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-9585-7993","authenticated-orcid":false,"given":"Miao","family":"Yu","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China and Squirrel Ai Learning, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7404-6748","authenticated-orcid":false,"given":"Fanci","family":"Meng","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China and Squirrel Ai Learning, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5011-756X","authenticated-orcid":false,"given":"Xinyun","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0734-6085","authenticated-orcid":false,"given":"Shilong","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China and Squirrel Ai Learning, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3450-0885","authenticated-orcid":false,"given":"Junyuan","family":"Mao","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, Anhui, China and Squirrel Ai Learning, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4784-9795","authenticated-orcid":false,"given":"Linsey","family":"Pan","sequence":"additional","affiliation":[{"name":"Salesforce, San Francisco, Illinois, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7774-8197","authenticated-orcid":false,"given":"Tianlong","family":"Chen","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0602-169X","authenticated-orcid":false,"given":"Kun","family":"Wang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9686-4369","authenticated-orcid":false,"given":"Xinfeng","family":"Li","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2633-8555","authenticated-orcid":false,"given":"Yongfeng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Rutgers University, New Brunswic, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7064-7438","authenticated-orcid":false,"given":"Bo","family":"An","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4516-2524","authenticated-orcid":false,"given":"Qingsong","family":"Wen","sequence":"additional","affiliation":[{"name":"Squirrel AI Learning, Seattle, Washington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Divyansh Agarwal Alexander R Fabbri Ben Risher Philippe Laban Shafiq Joty and Chien-Sheng Wu. 2024. Prompt Leakage effect and defense strategies for multi-turn LLM interactions. arXiv preprint arXiv:2404.16251(2024).","DOI":"10.18653\/v1\/2024.emnlp-industry.94"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Alfonso Amayuelas Xianjun Yang Antonis Antoniades Wenyue Hua Liangming Pan and William Wang. 2024. Multiagent collaboration attack: Investigating adversarial attacks in large language model collaborations via debate. arXiv preprint arXiv:2406.14711(2024).","DOI":"10.18653\/v1\/2024.findings-emnlp.407"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Maya Anderson Guy Amit and Abigail Goldsteen. 2024. Is my data in your retrieval database? membership inference attacks against retrieval augmented generation. arXiv preprint arXiv:2405.20446(2024).","DOI":"10.5220\/0013108300003899"},{"key":"e_1_3_2_2_4_1","volume-title":"Agentharm: A benchmark for measuring harmfulness of llm agents. arXiv preprint arXiv:2410.09024(2024).","author":"Andriushchenko Maksym","year":"2024","unstructured":"Maksym Andriushchenko, Alexandra Souly, Mateusz Dziemian, Derek Duenas, Maxwell Lin, Justin Wang, Dan Hendrycks, Andy Zou, Zico Kolter, Matt Fredrikson, et al., 2024. Agentharm: A benchmark for measuring harmfulness of llm agents. arXiv preprint arXiv:2410.09024(2024)."},{"key":"e_1_3_2_2_5_1","unstructured":"Md Ahsan Ayub and Subhabrata Majumdar. 2024. Embedding-based classifiers can detect prompt injection attacks. arXiv preprint arXiv:2410.22284(2024)."},{"key":"e_1_3_2_2_6_1","unstructured":"Eugene Bagdasaryan Tsung-Yin Hsieh Ben Nassi and Vitaly Shmatikov. 2023. Abusing images and sounds for indirect instruction injection in multi-modal LLMs. arXiv preprint arXiv:2307.10490(2023)."},{"key":"e_1_3_2_2_7_1","unstructured":"JUDGE BENCHMARK. [n.d.]. JAILJUDGE: AComprehensive JAILBREAK JUDGE BENCHMARK WITH MULTI-AGENT ENHANCED EXPLANATION EVALUATION FRAMEWORK. ( [n. d.])."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3674399.3674445"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3607199.3607237"},{"key":"e_1_3_2_2_11_1","unstructured":"Jizhou Chen and Samuel Lee Cong. 2025. AgentGuard: Repurposing Agentic Orchestrator for Safety Evaluation of Tool Orchestration. arXiv preprint arXiv:2502.09809(2025)."},{"key":"e_1_3_2_2_12_1","volume-title":"Struq: Defending against prompt injection with structured queries. arXiv preprint arXiv:2402.06363(2024).","author":"Chen Sizhe","year":"2024","unstructured":"Sizhe Chen, Julien Piet, Chawin Sitawarin, and David Wagner. 2024b. Struq: Defending against prompt injection with structured queries. arXiv preprint arXiv:2402.06363(2024)."},{"key":"e_1_3_2_2_13_1","first-page":"130185","article-title":"Agentpoison: Red-teaming llm agents via poisoning memory or knowledge bases","volume":"37","author":"Chen Zhaorun","year":"2025","unstructured":"Zhaorun Chen, Zhen Xiang, Chaowei Xiao, Dawn Song, and Bo Li. 2025. Agentpoison: Red-teaming llm agents via poisoning memory or knowledge bases. Advances in Neural Information Processing Systems, Vol. 37 (2025), 130185-130213.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_14_1","volume-title":"ICLR 2024 Workshop on Secure and Trustworthy Large Language Models.","author":"Chen Zhaorun","year":"2024","unstructured":"Zhaorun Chen, Zhuokai Zhao, Wenjie Qu, Zichen Wen, Zhiguang Han, Zhihong Zhu, Jiaheng Zhang, and Huaxiu Yao. 2024c. Pandora: Detailed llm jailbreaking via collaborated phishing agents with decomposed reasoning. In ICLR 2024 Workshop on Secure and Trustworthy Large Language Models."},{"key":"e_1_3_2_2_15_1","unstructured":"Wen Cheng Ke Sun Xinyu Zhang and Wei Wang. 2024b. Security Attacks on LLM-based Code Completion Tools. arXiv preprint arXiv:2408.11006(2024)."},{"key":"e_1_3_2_2_16_1","unstructured":"Yixin Cheng Markos Georgopoulos Volkan Cevher and Grigorios G Chrysos. 2024a. Leveraging the context through multi-round interactions for jailbreaking attacks. arXiv preprint arXiv:2402.09177(2024)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Sukmin Cho Soyeong Jeong Jeongyeon Seo Taeho Hwang and Jong C Park. 2024. Typos that Broke the RAG's Back: Genetic Attack on RAG Pipeline by Simulating Documents in the Wild via Low-level Perturbations. arXiv preprint arXiv:2404.13948(2024).","DOI":"10.18653\/v1\/2024.findings-emnlp.161"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Gobinda Chowdhury and Sudatta Chowdhury. 2024. AI-and LLM-driven search tools: A paradigm shift in information access for education and research. Journal of Information Science(2024) 01655515241284046.","DOI":"10.1177\/01655515241284046"},{"key":"e_1_3_2_2_19_1","volume-title":"Agentdojo: A dynamic environment to evaluate attacks and defenses for llm agents. arXiv preprint arXiv:2406.13352(2024).","author":"Debenedetti Edoardo","year":"2024","unstructured":"Edoardo Debenedetti, Jie Zhang, Mislav Balunovi\u0107, Luca Beurer-Kellner, Marc Fischer, and Florian Tram\u00e8r. 2024. Agentdojo: A dynamic environment to evaluate attacks and defenses for llm agents. arXiv preprint arXiv:2406.13352(2024)."},{"key":"e_1_3_2_2_20_1","first-page":"82895","article-title":"Agentdojo: A dynamic environment to evaluate prompt injection attacks and defenses for LLM agents","volume":"37","author":"Debenedetti Edoardo","year":"2025","unstructured":"Edoardo Debenedetti, Jie Zhang, Mislav Balunovic, Luca Beurer-Kellner, Marc Fischer, and Florian Tram\u00e8r. 2025. Agentdojo: A dynamic environment to evaluate prompt injection attacks and defenses for LLM agents. Advances in Neural Information Processing Systems, Vol. 37 (2025), 82895-82920.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_21_1","unstructured":"Zehang Deng Yongjian Guo Changzhou Han Wanlun Ma Junwu Xiong Sheng Wen and Yang Xiang. 2024. Ai agents under threat: A survey of key security challenges and future pathways. Comput. Surveys(2024)."},{"key":"e_1_3_2_2_22_1","volume-title":"l Segerie, and Vincent Corruble","author":"Dorn Diego","year":"2024","unstructured":"Diego Dorn, Alexandre Variengien, Charbel-Rapha AcG, l Segerie, and Vincent Corruble. 2024. Bells: A framework towards future proof benchmarks for the evaluation of llm safeguards. arXiv preprint arXiv:2406.01364(2024)."},{"key":"e_1_3_2_2_23_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Du Yilun","year":"2023","unstructured":"Yilun Du, Shuang Li, Antonio Torralba, Joshua B Tenenbaum, and Igor Mordatch. 2023. Improving factuality and reasoning in language models through multiagent debate. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_2_24_1","unstructured":"Richard Fang Rohan Bindu Akul Gupta Qiusi Zhan and Daniel Kang. 2024. Llm agents can autonomously hack websites. arXiv preprint arXiv:2402.06664(2024)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Ivar Frisch and Mario Giulianelli. 2024. LLM agents in interaction: Measuring personality consistency and linguistic alignment in interacting populations of large language models. arXiv preprint arXiv:2402.02896(2024).","DOI":"10.18653\/v1\/2024.personalize-1.9"},{"key":"e_1_3_2_2_26_1","volume-title":"Imprompter: Tricking LLM Agents into Improper Tool Use. arXiv preprint arXiv:2410.14923(2024).","author":"Fu Xiaohan","year":"2024","unstructured":"Xiaohan Fu, Shuheng Li, Zihan Wang, Yihao Liu, Rajesh K Gupta, Taylor Berg-Kirkpatrick, and Earlence Fernandes. 2024. Imprompter: Tricking LLM Agents into Improper Tool Use. arXiv preprint arXiv:2410.14923(2024)."},{"key":"e_1_3_2_2_27_1","unstructured":"Xiaohan Fu Zihan Wang Shuheng Li Rajesh K Gupta Niloofar Mireshghallah Taylor Berg-Kirkpatrick and Earlence Fernandes. 2023. Misusing tools in large language models with visual adversarial examples. arXiv preprint arXiv:2310.03185(2023)."},{"key":"e_1_3_2_2_28_1","unstructured":"Yuyou Gan Yong Yang Zhe Ma Ping He Rui Zeng Yiming Wang Qingming Li Chunyi Zhou Songze Li Ting Wang et al. 2024. Navigating the risks: A survey of security privacy and ethics threats in llm-based agents. arXiv preprint arXiv:2411.09523(2024)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Ge Gao Alexey Taymanov Eduardo Salinas Paul Mineiro and Dipendra Misra. 2024. Aligning llm agents by learning latent preference from user edits. arXiv preprint arXiv:2404.15269(2024).","DOI":"10.52202\/079017-4349"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3605764.3623985"},{"key":"e_1_3_2_2_31_1","unstructured":"Xiangming Gu Xiaosen Zheng Tianyu Pang Chao Du Qian Liu Ye Wang Jing Jiang and Min Lin. 2024. Agent smith: A single image can jailbreak one million multimodal llm agents exponentially fast. arXiv preprint arXiv:2402.08567(2024)."},{"key":"e_1_3_2_2_32_1","unstructured":"Taicheng Guo Xiuying Chen Yaqi Wang Ruidi Chang Shichao Pei Nitesh V Chawla Olaf Wiest and Xiangliang Zhang. 2024. Large language model based multi-agents: A survey of progress and challenges. arXiv preprint arXiv:2402.01680(2024)."},{"key":"e_1_3_2_2_33_1","unstructured":"Lewis Hammond Alan Chan Jesse Clifton Jason Hoelscher-Obermaier Akbir Khan Euan McLean Chandler Smith Wolfram Barfuss Jakob Foerster Tom\u00e1\u0161 Gaven\u010diak et al. 2025. Multi-Agent Risks from Advanced AI. arXiv preprint arXiv:2502.14143(2025)."},{"key":"e_1_3_2_2_34_1","unstructured":"Feng He Tianqing Zhu Dayong Ye Bo Liu Wanlei Zhou and Philip S Yu. 2024. The emerged security and privacy of llm agent: A survey with case studies. arXiv preprint arXiv:2407.19354(2024)."},{"key":"e_1_3_2_2_35_1","unstructured":"Pengfei He Yupin Lin Shen Dong Han Xu Yue Xing and Hui Liu. 2025. Red-Teaming LLM Multi-Agent Systems via Communication Attacks. arXiv preprint arXiv:2502.14847(2025)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.585"},{"key":"e_1_3_2_2_37_1","volume-title":"Trustllm: Trustworthiness in large language models. arXiv preprint arXiv:2401.05561(2024).","author":"Huang Yue","year":"2024","unstructured":"Yue Huang, Lichao Sun, Haoran Wang, Siyuan Wu, Qihui Zhang, Yuan Li, Chujie Gao, Yixin Huang, Wenhan Lyu, Yixuan Zhang, et al., 2024. Trustllm: Trustworthiness in large language models. arXiv preprint arXiv:2401.05561(2024)."},{"key":"e_1_3_2_2_38_1","volume-title":"Enhancing Fake News Detection with Large Language Models Through Multi-agent Debates. In CCF International Conference on Natural Language Processing and Chinese Computing. Springer, 474-486","author":"Jeptoo Korir Nancy","year":"2024","unstructured":"Korir Nancy Jeptoo and Chengjie Sun. 2024. Enhancing Fake News Detection with Large Language Models Through Multi-agent Debates. In CCF International Conference on Natural Language Processing and Chinese Computing. Springer, 474-486."},{"key":"e_1_3_2_2_39_1","volume-title":"Rag-thief: Scalable extraction of private data from retrieval-augmented generation applications with agent-based attacks. arXiv preprint arXiv:2411.14110(2024).","author":"Jiang Changyue","year":"2024","unstructured":"Changyue Jiang, Xudong Pan, Geng Hong, Chenfu Bao, and Min Yang. 2024. Rag-thief: Scalable extraction of private data from retrieval-augmented generation applications with agent-based attacks. arXiv preprint arXiv:2411.14110(2024)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Ziyou Jiang Mingyang Li Guowei Yang Junjie Wang Yuekai Huang Zhiyuan Chang and Qing Wang. 2025. Mimicking the Familiar: Dynamic Command Generation for Information Theft Attacks in LLM Tool-Learning System. arXiv preprint arXiv:2502.11358(2025).","DOI":"10.18653\/v1\/2025.acl-long.672"},{"key":"e_1_3_2_2_41_1","unstructured":"Tianjie Ju Yiting Wang Xinbei Ma Pengzhou Cheng Haodong Zhao Yulong Wang Lifeng Liu Jian Xie Zhuosheng Zhang and Gongshen Liu. 2024. Flooding spread of manipulated knowledge in llm-based multi-agent communities. arXiv preprint arXiv:2407.07791(2024)."},{"key":"e_1_3_2_2_42_1","volume-title":"Soheil Feizi, and Himabindu Lakkaraju.","author":"Kumar Aounon","year":"2023","unstructured":"Aounon Kumar, Chirag Agarwal, Suraj Srinivas, Aaron Jiaxun Li, Soheil Feizi, and Himabindu Lakkaraju. 2023. Certifying llm safety against adversarial prompting. arXiv preprint arXiv:2309.02705(2023)."},{"key":"e_1_3_2_2_43_1","volume-title":"Elaine Chang, Vaughn Robinson, Sean Hendryx, Shuyan Zhou, Matt Fredrikson, et al.","author":"Kumar Priyanshu","year":"2024","unstructured":"Priyanshu Kumar, Elaine Lau, Saranya Vijayakumar, Tu Trinh, Scale Red Team, Elaine Chang, Vaughn Robinson, Sean Hendryx, Shuyan Zhou, Matt Fredrikson, et al., 2024. Refusal-trained llms are easily jailbroken as browser agents. arXiv preprint arXiv:2410.13886(2024)."},{"key":"e_1_3_2_2_44_1","unstructured":"Ted Kwartler Matthew Berman and Alan Aqrawi. 2024. Good Parenting is all you need-Multi-agentic LLM Hallucination Mitigation. arXiv preprint arXiv:2410.14262(2024)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Ohjoon Kwon Donghyeon Jeon Nayoung Choi Gyu-Hwung Cho Changbong Kim Hyunwoo Lee Inho Kang Sun Kim and Taiwoo Park. 2024. SLM as Guardian: Pioneering AI Safety with Small Language Models. arXiv preprint arXiv:2405.19795(2024).","DOI":"10.18653\/v1\/2024.emnlp-industry.99"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Lucio La Cava and Andrea Tagarelli. 2024. Safeguarding Decentralized Social Media: LLM Agents for Automating Community Rule Compliance. arXiv preprint arXiv:2409.08963(2024).","DOI":"10.1016\/j.osnem.2025.100319"},{"key":"e_1_3_2_2_47_1","unstructured":"Donghyun Lee and Mo Tiwari. 2024. Prompt infection: Llm-to-llm prompt injection within multi-agent systems. arXiv preprint arXiv:2410.07283(2024)."},{"key":"e_1_3_2_2_48_1","volume-title":"Tom Goldstein, and Micah Goldblum.","author":"Li Ang","year":"2025","unstructured":"Ang Li, Yin Zhou, Vethavikashini Chithrra Raghuram, Tom Goldstein, and Micah Goldblum. 2025. Commercial LLM Agents Are Already Vulnerable to Simple Yet Dangerous Attacks. arXiv preprint arXiv:2502.08586(2025)."},{"key":"e_1_3_2_2_49_1","first-page":"51991","article-title":"Camel: Communicative agents for ''mind'' exploration of large language model society","volume":"36","author":"Li Guohao","year":"2023","unstructured":"Guohao Li, Hasan Hammoud, Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem. 2023a. Camel: Communicative agents for ''mind'' exploration of large language model society. Advances in Neural Information Processing Systems, Vol. 36 (2023), 51991-52008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_50_1","unstructured":"Haoran Li Mingshi Xu and Yangqiu Song. 2023b. Sentence embedding leaks more information than you expect: Generative embedding inversion attack to recover the whole sentence. arXiv preprint arXiv:2305.03010(2023)."},{"key":"e_1_3_2_2_51_1","unstructured":"Yuanchun Li Hao Wen Weijun Wang Xiangyu Li Yizhen Yuan Guohong Liu Jiacheng Liu Wenxing Xu Xiang Wang Yi Sun et al. 2024. Personal llm agents: Insights and survey about the capability efficiency and security. arXiv preprint arXiv:2401.05459(2024)."},{"key":"e_1_3_2_2_52_1","unstructured":"Guang Lin and Qibin Zhao. 2024. Large Language Model Sentinel: LLM Agent for Adversarial Purification. arXiv preprint arXiv:2405.20770(2024)."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.267"},{"key":"e_1_3_2_2_54_1","unstructured":"Xiaogeng Liu Zhiyuan Yu Yizhe Zhang Ning Zhang and Chaowei Xiao. 2024b. Automatic and universal prompt injection attacks against large language models. arXiv preprint arXiv:2403.04957(2024)."},{"key":"e_1_3_2_2_55_1","volume-title":"Yegor Klochkov, Muhammad Faaiz Taufiq, and Hang Li.","author":"Liu Yang","year":"2023","unstructured":"Yang Liu, Yuanshun Yao, Jean-Francois Ton, Xiaoying Zhang, Ruocheng Guo Hao Cheng, Yegor Klochkov, Muhammad Faaiz Taufiq, and Hang Li. 2023. Trustworthy LLMs: A survey and guideline for evaluating large language models' alignment. arXiv preprint arXiv:2308.05374(2023)."},{"key":"e_1_3_2_2_56_1","unstructured":"Zihan Liu Ruinan Zeng Dongxia Wang Gengyun Peng Jingyi Wang Qiang Liu Peiyu Liu and Wenhai Wang. 2024c. Agents4PLC: Automating Closed-loop PLC Code Generation and Verification in Industrial Control Systems using LLM-based Agents. arXiv preprint arXiv:2410.14209(2024)."},{"key":"e_1_3_2_2_57_1","unstructured":"Tula Masterman Sandi Besen Mason Sawtell and Alex Chao. 2024. The landscape of emerging ai agent architectures for reasoning planning and tool calling: A survey. arXiv preprint arXiv:2404.11584(2024)."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"crossref","unstructured":"John X Morris Volodymyr Kuleshov Vitaly Shmatikov and Alexander M Rush. 2023. Text embeddings reveal (almost) as much as text. arXiv preprint arXiv:2310.06816(2023).","DOI":"10.18653\/v1\/2023.emnlp-main.765"},{"key":"e_1_3_2_2_59_1","volume-title":"Neel Kant, Kriti Aggarwal, Neha Manjunath, Debajyoti Datta, Zhengliang Liu, Jiayuan Ding, Sophia Busacca, et al.","author":"Mukherjee Subhabrata","year":"2024","unstructured":"Subhabrata Mukherjee, Paul Gamble, Markel Sanz Ausin, Neel Kant, Kriti Aggarwal, Neha Manjunath, Debajyoti Datta, Zhengliang Liu, Jiayuan Ding, Sophia Busacca, et al., 2024. Polaris: A safety-focused llm constellation architecture for healthcare. arXiv preprint arXiv:2403.13313(2024)."},{"key":"e_1_3_2_2_60_1","unstructured":"Yuzhou Nie Zhun Wang Ye Yu Xian Wu Xuandong Zhao Wenbo Guo and Dawn Song. 2024. PrivAgent: Agentic-based Red-teaming for LLM Privacy Leakage. arXiv preprint arXiv:2412.05734(2024)."},{"key":"e_1_3_2_2_61_1","volume-title":"Cooperative multi-agent learning: The state of the art. Autonomous agents and multi-agent systems","author":"Panait Liviu","year":"2005","unstructured":"Liviu Panait and Sean Luke. 2005. Cooperative multi-agent learning: The state of the art. Autonomous agents and multi-agent systems, Vol. 11 (2005), 387-434."},{"key":"e_1_3_2_2_62_1","volume-title":"ICLR 2024 Workshop on Large Language Model (LLM) Agents.","author":"Pang Xianghe","year":"2024","unstructured":"Xianghe Pang, Shuo Tang, Rui Ye, Yuxin Xiong, Bolun Zhang, Yanfeng Wang, and Siheng Chen. 2024. Self-alignment of large language models via multi-agent social simulation. In ICLR 2024 Workshop on Large Language Model (LLM) Agents."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3704435"},{"key":"e_1_3_2_2_64_1","unstructured":"Yangjun Ruan Honghua Dong Andrew Wang Silviu Pitis Yongchao Zhou Jimmy Ba Yann Dubois Chris J Maddison and Tatsunori Hashimoto. 2023. Identifying the risks of lm agents with an lm-emulated sandbox. arXiv preprint arXiv:2309.15817(2023)."},{"key":"e_1_3_2_2_65_1","unstructured":"Mark Russinovich Ahmed Salem and Ronen Eldan. 2024. Great now write an article about that: The crescendo multi-turn llm jailbreak attack. arXiv preprint arXiv:2404.01833(2024)."},{"key":"e_1_3_2_2_66_1","unstructured":"Zhuocheng Shen. 2024. Llm with tools: A survey. arXiv preprint arXiv:2409.18807(2024)."},{"key":"e_1_3_2_2_67_1","unstructured":"Chengyu Song Linru Ma Jianming Zheng Jinzhi Liao Hongyu Kuang and Lin Yang. 2024a. Audit-LLM: Multi-Agent Collaboration for Log-based Insider Threat Detection. arXiv preprint arXiv:2408.08902(2024)."},{"key":"e_1_3_2_2_68_1","volume-title":"Hyungsub Kim, Antonio Bianchi, and Z Berkay Celik.","author":"Song Ruoyu","year":"2024","unstructured":"Ruoyu Song, Muslum Ozgur Ozmen, Hyungsub Kim, Antonio Bianchi, and Z Berkay Celik. 2024b. Enhancing llm-based autonomous driving agents to mitigate perception attacks. arXiv preprint arXiv:2409.14488(2024)."},{"key":"e_1_3_2_2_69_1","unstructured":"Zhen Tan Chengshuai Zhao Raha Moraffah Yifan Li Yu Kong Tianlong Chen and Huan Liu. 2024. The wolf within: Covert injection of malice into mllm societies via an mllm operative. arXiv preprint arXiv:2402.14859(2024)."},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"crossref","unstructured":"Xiangru Tang Qiao Jin Kunlun Zhu Tongxin Yuan Yichi Zhang Wangchunshu Zhou Meng Qu Yilun Zhao Jian Tang Zhuosheng Zhang et al. 2024. Prioritizing safeguarding over autonomy: Risks of llm agents for science. arXiv preprint arXiv:2402.04247(2024).","DOI":"10.1038\/s41467-025-63913-1"},{"key":"e_1_3_2_2_71_1","unstructured":"Elizaveta Tennant Stephen Hailes and Mirco Musolesi. 2024. Moral Alignment for LLM Agents. arXiv preprint arXiv:2410.01639(2024)."},{"key":"e_1_3_2_2_72_1","unstructured":"Yu Tian Xiao Yang Jingyuan Zhang Yinpeng Dong and Hang Su. 2023. Evil geniuses: Delving into the safety of llm-based agents. arXiv preprint arXiv:2311.11855(2023)."},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"crossref","unstructured":"Terry Tong Jiashu Xu Qin Liu and Muhao Chen. 2024. Securing Multi-turn Conversational Language Models From Distributed Backdoor Triggers. arXiv preprint arXiv:2407.04151(2024).","DOI":"10.18653\/v1\/2024.findings-emnlp.750"},{"key":"e_1_3_2_2_74_1","unstructured":"Javal Vyas and Mehmet Mercang\u00f6z. 2024. Autonomous Industrial Control using an Agentic Framework with Large Language Models. arXiv preprint arXiv:2411.05904(2024)."},{"key":"e_1_3_2_2_75_1","volume-title":"Mrj-agent: An effective jailbreak agent for multi-round dialogue. arXiv preprint arXiv:2411.03814(2024).","author":"Wang Fengxiang","year":"2024","unstructured":"Fengxiang Wang, Ranjie Duan, Peng Xiao, Xiaojun Jia, YueFeng Chen, Chongwen Wang, Jialing Tao, Hang Su, Jun Zhu, and Hui Xue. 2024a. Mrj-agent: An effective jailbreak agent for multi-round dialogue. arXiv preprint arXiv:2411.03814(2024)."},{"key":"e_1_3_2_2_76_1","unstructured":"Haowei Wang Rupeng Zhang Junjie Wang Mingyang Li Yuekai Huang Dandan Wang and Qing Wang. 2024 e. From Allies to Adversaries: Manipulating LLM Tool-Calling through Adversarial Injection. arXiv preprint arXiv:2412.10198(2024)."},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-024-40231-1"},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"crossref","unstructured":"Shilong Wang Guibin Zhang Miao Yu Guancheng Wan Fanci Meng Chongye Guo Kun Wang and Yang Wang. 2025. G-Safeguard: A Topology-Guided Security Lens and Treatment on LLM-based Multi-agent Systems. arXiv preprint arXiv:2502.11127(2025).","DOI":"10.18653\/v1\/2025.acl-long.359"},{"key":"e_1_3_2_2_79_1","unstructured":"Yuntao Wang Yanghe Pan Quan Zhao Yi Deng Zhou Su Linkang Du and Tom H Luan. 2024c. Large Model Agents: State-of-the-Art Cooperation Paradigms Security and Privacy and Future Trends. arXiv preprint arXiv:2409.14457(2024)."},{"key":"e_1_3_2_2_80_1","volume-title":"Badagent: Inserting and activating backdoor attacks in llm agents. arXiv preprint arXiv:2406.03007(2024).","author":"Wang Yifei","year":"2024","unstructured":"Yifei Wang, Dizhan Xue, Shengjie Zhang, and Shengsheng Qian. 2024d. Badagent: Inserting and activating backdoor attacks in llm agents. arXiv preprint arXiv:2406.03007(2024)."},{"key":"e_1_3_2_2_81_1","volume-title":"Autogen: Enabling next-gen llm applications via multi-agent conversation framework. arXiv preprint arXiv:2308.08155(2023).","author":"Wu Qingyun","year":"2023","unstructured":"Qingyun Wu, Gagan Bansal, Jieyu Zhang, Yiran Wu, Shaokun Zhang, Erkang Zhu, Beibin Li, Li Jiang, Xiaoyun Zhang, and Chi Wang. 2023. Autogen: Enabling next-gen llm applications via multi-agent conversation framework. arXiv preprint arXiv:2308.08155(2023)."},{"key":"e_1_3_2_2_82_1","volume-title":"SELP: Generating safe and efficient task plans for robot agents with large language models. arXiv preprint arXiv:2409.19471(2024).","author":"Wu Yi","year":"2024","unstructured":"Yi Wu, Zikang Xiong, Yiran Hu, Shreyash S Iyengar, Nan Jiang, Aniket Bera, Lin Tan, and Suresh Jagannathan. 2024. SELP: Generating safe and efficient task plans for robot agents with large language models. arXiv preprint arXiv:2409.19471(2024)."},{"key":"e_1_3_2_2_83_1","unstructured":"Xun Xian Ganghua Wang Xuan Bi Jayanth Srinivasa Ashish Kundu Charles Fleming Mingyi Hong and Jie Ding. 2024. On the Vulnerability of Applying Retrieval-Augmented Generation within Knowledge-Intensive Application Domains. arXiv preprint arXiv:2409.17275(2024)."},{"key":"e_1_3_2_2_84_1","unstructured":"Chong Xiang Tong Wu Zexuan Zhong David Wagner Danqi Chen and Prateek Mittal. 2024a. Certifiably robust rag against retrieval corruption. arXiv preprint arXiv:2405.15556(2024)."},{"key":"e_1_3_2_2_85_1","volume-title":"Guardagent: Safeguard llm agents by a guard agent via knowledge-enabled reasoning. arXiv preprint arXiv:2406.09187(2024).","author":"Xiang Zhen","year":"2024","unstructured":"Zhen Xiang, Linzhi Zheng, Yanjie Li, Junyuan Hong, Qinbin Li, Han Xie, Jiawei Zhang, Zidi Xiong, Chulin Xie, Carl Yang, et al., 2024b. Guardagent: Safeguard llm agents by a guard agent via knowledge-enabled reasoning. arXiv preprint arXiv:2406.09187(2024)."},{"key":"e_1_3_2_2_86_1","volume-title":"Redagent: Red teaming large language models with context-aware autonomous language agent. arXiv preprint arXiv:2407.16667(2024).","author":"Xu Huiyu","year":"2024","unstructured":"Huiyu Xu, Wenhui Zhang, Zhibo Wang, Feng Xiao, Rui Zheng, Yunhe Feng, Zhongjie Ba, and Kui Ren. 2024. Redagent: Red teaming large language models with context-aware autonomous language agent. arXiv preprint arXiv:2407.16667(2024)."},{"key":"e_1_3_2_2_87_1","unstructured":"Wenkai Yang Xiaohan Bi Yankai Lin Sishuo Chen Jie Zhou and Xu Sun. 2024a. Watch out for your agents! investigating backdoor threats to llm-based agents. arXiv preprint arXiv:2402.11208(2024)."},{"key":"e_1_3_2_2_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02482"},{"key":"e_1_3_2_2_89_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611447"},{"key":"e_1_3_2_2_90_1","volume-title":"Toolsword: Unveiling safety issues of large language models in tool learning across three stages. arXiv preprint arXiv:2402.10753(2024).","author":"Ye Junjie","year":"2024","unstructured":"Junjie Ye, Sixian Li, Guanyu Li, Caishuang Huang, Songyang Gao, Yilong Wu, Qi Zhang, Tao Gui, and Xuanjing Huang. 2024. Toolsword: Unveiling safety issues of large language models in tool learning across three stages. arXiv preprint arXiv:2402.10753(2024)."},{"key":"e_1_3_2_2_91_1","unstructured":"Sheng Yin Xianghe Pang Yuanzhuo Ding Menglan Chen Yutong Bi Yichen Xiong Wenhao Huang Zhen Xiang Jing Shao and Siheng Chen. 2024. SafeAgentBench: A Benchmark for Safe Task Planning of Embodied LLM Agents. arXiv preprint arXiv:2412.13178(2024)."},{"key":"e_1_3_2_2_92_1","volume-title":"Netsafe: Exploring the topological safety of multi-agent networks. arXiv preprint arXiv:2410.15686(2024).","author":"Yu Miao","year":"2024","unstructured":"Miao Yu, Shilong Wang, Guibin Zhang, Junyuan Mao, Chenlong Yin, Qijiong Liu, Qingsong Wen, Kun Wang, and Yang Wang. 2024. Netsafe: Exploring the topological safety of multi-agent networks. arXiv preprint arXiv:2410.15686(2024)."},{"key":"e_1_3_2_2_93_1","volume-title":"BLAST: A Stealthy Backdoor Leverage Attack against Cooperative Multi-Agent Deep Reinforcement Learning based Systems. arXiv preprint arXiv:2501.01593(2025).","author":"Yu Yinbo","year":"2025","unstructured":"Yinbo Yu, Saihao Yan, Xueyu Yin, Jing Fang, and Jiajia Liu. 2025. BLAST: A Stealthy Backdoor Leverage Attack against Cooperative Multi-Agent Deep Reinforcement Learning based Systems. arXiv preprint arXiv:2501.01593(2025)."},{"key":"e_1_3_2_2_94_1","volume-title":"R-judge: Benchmarking safety risk awareness for llm agents. arXiv preprint arXiv:2401.10019(2024).","author":"Yuan Tongxin","year":"2024","unstructured":"Tongxin Yuan, Zhiwei He, Lingzhong Dong, Yiming Wang, Ruijie Zhao, Tian Xia, Lizhen Xu, Binglin Zhou, Fangqi Li, Zhuosheng Zhang, et al., 2024a. R-judge: Benchmarking safety risk awareness for llm agents. arXiv preprint arXiv:2401.10019(2024)."},{"key":"e_1_3_2_2_95_1","volume-title":"S-eval: Automatic and adaptive test generation for benchmarking safety evaluation of large language models. arXiv preprint arXiv:2405.14191(2024).","author":"Yuan Xiaohan","year":"2024","unstructured":"Xiaohan Yuan, Jinfeng Li, Dongxia Wang, Yuefeng Chen, Xiaofeng Mao, Longtao Huang, Hui Xue, Wenhai Wang, Kui Ren, and Jingyi Wang. 2024b. S-eval: Automatic and adaptive test generation for benchmarking safety evaluation of large language models. arXiv preprint arXiv:2405.14191(2024)."},{"key":"e_1_3_2_2_96_1","doi-asserted-by":"crossref","unstructured":"Shenglai Zeng Jiankun Zhang Pengfei He Yue Xing Yiding Liu Han Xu Jie Ren Shuaiqiang Wang Dawei Yin Yi Chang et al. 2024b. The good and the bad: Exploring privacy issues in retrieval-augmented generation (rag). arXiv preprint arXiv:2402.16893(2024).","DOI":"10.18653\/v1\/2024.findings-acl.267"},{"key":"e_1_3_2_2_97_1","volume-title":"Autodefense: Multi-agent llm defense against jailbreak attacks. arXiv preprint arXiv:2403.04783(2024).","author":"Zeng Yifan","year":"2024","unstructured":"Yifan Zeng, Yiran Wu, Xiao Zhang, Huazheng Wang, and Qingyun Wu. 2024a. Autodefense: Multi-agent llm defense against jailbreak attacks. arXiv preprint arXiv:2403.04783(2024)."},{"key":"e_1_3_2_2_98_1","volume-title":"Injecagent: Benchmarking indirect prompt injections in tool-integrated large language model agents. arXiv preprint arXiv:2403.02691(2024).","author":"Zhan Qiusi","year":"2024","unstructured":"Qiusi Zhan, Zhixiang Liang, Zifan Ying, and Daniel Kang. 2024. Injecagent: Benchmarking indirect prompt injections in tool-integrated large language model agents. arXiv preprint arXiv:2403.02691(2024)."},{"key":"e_1_3_2_2_99_1","doi-asserted-by":"crossref","unstructured":"Boyang Zhang Yicong Tan Yun Shen Ahmed Salem Michael Backes Savvas Zannettou and Yang Zhang. 2024 h. Breaking agents: Compromising autonomous llm agents through malfunction amplification. arXiv preprint arXiv:2407.20859(2024).","DOI":"10.18653\/v1\/2025.emnlp-main.1771"},{"key":"e_1_3_2_2_100_1","volume-title":"Ufo: A ui-focused agent for windows os interaction. arXiv preprint arXiv:2402.07939(2024).","author":"Zhang Chaoyun","year":"2024","unstructured":"Chaoyun Zhang, Liqun Li, Shilin He, Xu Zhang, Bo Qiao, Si Qin, Minghua Ma, Yu Kang, Qingwei Lin, Saravan Rajmohan, et al., 2024 f. Ufo: A ui-focused agent for windows os interaction. arXiv preprint arXiv:2402.07939(2024)."},{"key":"e_1_3_2_2_101_1","unstructured":"Guibin Zhang Luyang Niu Junfeng Fang Kun Wang Lei Bai and Xiang Wang. 2025. Multi-agent Architecture Search via Agentic Supernet. arXiv preprint arXiv:2502.04180(2025)."},{"key":"e_1_3_2_2_102_1","unstructured":"Hanrong Zhang Jingyuan Huang Kai Mei Yifei Yao Zhenting Wang Chenlu Zhan Hongwei Wang and Yongfeng Zhang. 2024 e. Agent security bench (asb): Formalizing and benchmarking attacks and defenses in llm-based agents. arXiv preprint arXiv:2410.02644(2024)."},{"key":"e_1_3_2_2_103_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01464"},{"key":"e_1_3_2_2_104_1","doi-asserted-by":"publisher","DOI":"10.65109\/PXEC2992"},{"key":"e_1_3_2_2_105_1","unstructured":"Zeyu Zhang Xiaohe Bo Chen Ma Rui Li Xu Chen Quanyu Dai Jieming Zhu Zhenhua Dong and Ji-Rong Wen. 2024a. A survey on the memory mechanism of large language model based agents. arXiv preprint arXiv:2404.13501(2024)."},{"key":"e_1_3_2_2_106_1","unstructured":"Zhexin Zhang Shiyao Cui Yida Lu Jingzhuo Zhou Junxiao Yang Hongning Wang and Minlie Huang. 2024b. Agent-SafetyBench: Evaluating the Safety of LLM Agents. arXiv preprint arXiv:2412.14470(2024)."},{"key":"e_1_3_2_2_107_1","unstructured":"Zhexin Zhang Shiyao Cui Yida Lu Jingzhuo Zhou Junxiao Yang Hongning Wang and Minlie Huang. 2024c. Agent-SafetyBench: Evaluating the Safety of LLM Agents. arXiv preprint arXiv:2412.14470(2024)."},{"key":"e_1_3_2_2_108_1","volume-title":"Shieldlm: Empowering llms as aligned, customizable and explainable safety detectors. arXiv preprint arXiv:2402.16444(2024).","author":"Zhang Zhexin","year":"2024","unstructured":"Zhexin Zhang, Yida Lu, Jingyuan Ma, Di Zhang, Rui Li, Pei Ke, Hao Sun, Lei Sha, Zhifang Sui, Hongning Wang, et al., 2024 g. Shieldlm: Empowering llms as aligned, customizable and explainable safety detectors. arXiv preprint arXiv:2402.16444(2024)."},{"key":"e_1_3_2_2_109_1","doi-asserted-by":"crossref","unstructured":"Zaibin Zhang Yongting Zhang Lijun Li Hongzhi Gao Lijun Wang Huchuan Lu Feng Zhao Yu Qiao and Jing Shao. 2024 j. Psysafe: A comprehensive framework for psychological-based attack defense and evaluation of multi-agent system safety. arXiv preprint arXiv:2401.11880(2024).","DOI":"10.18653\/v1\/2024.acl-long.812"},{"key":"e_1_3_2_2_110_1","doi-asserted-by":"crossref","unstructured":"Zexuan Zhong Ziqing Huang Alexander Wettig and Danqi Chen. 2023. Poisoning retrieval corpora by injecting adversarial passages. arXiv preprint arXiv:2310.19156(2023).","DOI":"10.18653\/v1\/2023.emnlp-main.849"},{"key":"e_1_3_2_2_111_1","doi-asserted-by":"crossref","unstructured":"Huichi Zhou Kin-Hei Lee Zhonghao Zhan Yue Chen and Zhenhao Li. 2025a. TrustRAG: Enhancing Robustness and Trustworthiness in RAG. arXiv preprint arXiv:2501.00879(2025).","DOI":"10.32388\/Z4DWHQ"},{"key":"e_1_3_2_2_112_1","volume-title":"Yejin Choi, Niloofar Mireshghallah, et al.","author":"Zhou Xuhui","year":"2024","unstructured":"Xuhui Zhou, Hyunwoo Kim, Faeze Brahman, Liwei Jiang, Hao Zhu, Ximing Lu, Frank Xu, Bill Yuchen Lin, Yejin Choi, Niloofar Mireshghallah, et al., 2024. Haicosystem: An ecosystem for sandboxing safety risks in human-ai interactions. arXiv preprint arXiv:2409.16427(2024)."},{"key":"e_1_3_2_2_113_1","doi-asserted-by":"crossref","unstructured":"Yihe Zhou Tao Ni Wei-Bin Lee and Qingchuan Zhao. 2025c. A Survey on Backdoor Threats in Large Language Models (LLMs): Attacks Defenses and Evaluations. arXiv preprint arXiv:2502.05224(2025).","DOI":"10.53941\/tai.2025.100003"},{"key":"e_1_3_2_2_114_1","volume-title":"CORBA: Contagious Recursive Blocking Attacks on Multi-Agent Systems Based on Large Language Models. arXiv preprint arXiv:2502.14529(2025).","author":"Zhou Zhenhong","year":"2025","unstructured":"Zhenhong Zhou, Zherui Li, Jie Zhang, Yuanhe Zhang, Kun Wang, Yang Liu, and Qing Guo. 2025b. CORBA: Contagious Recursive Blocking Attacks on Multi-Agent Systems Based on Large Language Models. arXiv preprint arXiv:2502.14529(2025)."},{"key":"e_1_3_2_2_115_1","unstructured":"Pengyu Zhu Zhenhong Zhou Yuanhe Zhang Shilinlu Yan Kun Wang and Sen Su. 2025. DemonAgent: Dynamically Encrypted Multi-Backdoor Implantation Attack on LLM-based Agent. arXiv preprint arXiv:2502.12575(2025)."},{"key":"e_1_3_2_2_116_1","volume-title":"Riskawarebench: Towards evaluating physical risk awareness for high-level planning of llm-based embodied agents. arXiv e-prints(2024), arXiv-2408.","author":"Zhu Zihao","year":"2024","unstructured":"Zihao Zhu, Bingzhe Wu, Zhengyou Zhang, and Baoyuan Wu. 2024. Riskawarebench: Towards evaluating physical risk awareness for high-level planning of llm-based embodied agents. arXiv e-prints(2024), arXiv-2408."},{"key":"e_1_3_2_2_117_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Zhuge Mingchen","year":"2024","unstructured":"Mingchen Zhuge, Wenyi Wang, Louis Kirsch, Francesco Faccio, Dmitrii Khizbullin, and J\u00fcrgen Schmidhuber. 2024. Gptswarm: Language agents as optimizable graphs. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_2_118_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J Zico Kolter and Matt Fredrikson. 2023. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043(2023)."},{"key":"e_1_3_2_2_119_1","volume-title":"Poisonedrag: Knowledge corruption attacks to retrieval-augmented generation of large language models. arXiv preprint arXiv:2402.07867(2024).","author":"Zou Wei","year":"2024","unstructured":"Wei Zou, Runpeng Geng, Binghui Wang, and Jinyuan Jia. 2024. Poisonedrag: Knowledge corruption attacks to retrieval-augmented generation of large language models. arXiv preprint arXiv:2402.07867(2024)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3736561","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T18:01:13Z","timestamp":1777572073000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3736561"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":119,"alternative-id":["10.1145\/3711896.3736561","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3736561","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}