{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:18:58Z","timestamp":1783153138095,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","funder":[{"name":"National Key Research and Development Program of China","award":["2024YFF0618800"],"award-info":[{"award-number":["2024YFF0618800"]}]},{"name":"National Natural Science Foundation of China","award":["62402114"],"award-info":[{"award-number":["62402114"]}]},{"name":"Shanghai Municipal Education Commission","award":["24KXZNA08"],"award-info":[{"award-number":["24KXZNA08"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792462","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T12:38:33Z","timestamp":1777293513000},"page":"3135-3146","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SentinelNet: Safeguarding Multi-Agent Collaboration Through Credit-Based Dynamic Threat Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5064-1892","authenticated-orcid":false,"given":"Yang","family":"Feng","sequence":"first","affiliation":[{"name":"The University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1394-0395","authenticated-orcid":false,"given":"Xudong","family":"Pan","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China and Shanghai Innovation Institute, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"When persuasion overrides truth in multi-agent llm debates: Introducing a confidence-weighted persuasion override rate (cw-por). arXiv preprint arXiv:2504.00374","author":"Agarwal Mahak","year":"2025","unstructured":"Mahak Agarwal and Divyam Khanna. 2025. When persuasion overrides truth in multi-agent llm debates: Introducing a confidence-weighted persuasion override rate (cw-por). arXiv preprint arXiv:2504.00374 (2025)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.407"},{"key":"e_1_3_2_1_3_1","unstructured":"Anthropic. 2024. Claude Code. https:\/\/claude.com\/product\/claude-code. Accessed: 2025-10-04."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12243-020-00751-w"},{"key":"e_1_3_2_1_5_1","volume-title":"Combating adversarial attacks with multi-agent debate. arXiv preprint arXiv:2401.05998","author":"Chern Steffi","year":"2024","unstructured":"Steffi Chern, Zhen Fan, and Andy Liu. 2024. Combating adversarial attacks with multi-agent debate. arXiv preprint arXiv:2401.05998 (2024)."},{"key":"e_1_3_2_1_6_1","unstructured":"Karl Cobbe Vineet Kosaraju Mohammad Bavarian Mark Chen Heewoo Jun Lukasz Kaiser Matthias Plappert Jerry Tworek Jacob Hilton Reiichiro Nakano et al. 2021. Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)."},{"key":"e_1_3_2_1_7_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Du Yilun","year":"2023","unstructured":"Yilun Du, Shuang Li, Antonio Torralba, Joshua B Tenenbaum, and Igor Mordatch. 2023. Improving factuality and reasoning in language models through multiagent debate. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.3390\/app12199901"},{"key":"e_1_3_2_1_9_1","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et al. 2024. A survey on llm-as-a-judge. arXiv preprint arXiv:2411.15594 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Brandon Waldon, Daniel Rockmore, Diego Zambrano, et al.","author":"Guha Neel","year":"2023","unstructured":"Neel Guha, Julian Nyarko, Daniel Ho, Christopher R\u00e9, Adam Chilton, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel Rockmore, Diego Zambrano, et al., 2023. Legalbench: A collaboratively built benchmark for measuring legal reasoning in large language models. Advances in neural information processing systems, Vol. 36 (2023), 44123-44279."},{"key":"e_1_3_2_1_11_1","volume-title":"Trust: Attention-based Trust Management for LLM Multi-Agent Systems. arXiv preprint arXiv:2506.02546","author":"He Pengfei","year":"2025","unstructured":"Pengfei He, Zhenwei Dai, Xianfeng Tang, Yue Xing, Hui Liu, Jingying Zeng, Qiankun Peng, Shrivats Agrawal, Samarth Varshney, Suhang Wang, et al., 2025a. Attention Knows Whom to Trust: Attention-based Trust Management for LLM Multi-Agent Systems. arXiv preprint arXiv:2506.02546 (2025)."},{"key":"e_1_3_2_1_12_1","volume-title":"Red-teaming llm multi-agent systems via communication attacks. arXiv preprint arXiv:2502.14847","author":"He Pengfei","year":"2025","unstructured":"Pengfei He, Yupin Lin, Shen Dong, Han Xu, Yue Xing, and Hui Liu. 2025b. Red-teaming llm multi-agent systems via communication attacks. arXiv preprint arXiv:2502.14847 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Measuring Massive Multitask Language Understanding. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=d7KBjmI3GmQ","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2021. Measuring Massive Multitask Language Understanding. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=d7KBjmI3GmQ"},{"key":"e_1_3_2_1_14_1","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et al., 2022. Lora: Low-rank adaptation of large language models. ICLR, Vol. 1, 2 (2022), 3.","journal-title":"ICLR"},{"key":"e_1_3_2_1_15_1","volume-title":"Trustagent: Towards safe and trustworthy llm-based agents. arXiv preprint arXiv:2402.01586","author":"Hua Wenyue","year":"2024","unstructured":"Wenyue Hua, Xianjun Yang, Mingyu Jin, Zelong Li, Wei Cheng, Ruixiang Tang, and Yongfeng Zhang. 2024. Trustagent: Towards safe and trustworthy llm-based agents. arXiv preprint arXiv:2402.01586 (2024)."},{"key":"e_1_3_2_1_16_1","volume-title":"On the resilience of llm-based multi-agent collaboration with faulty agents. arXiv preprint arXiv:2408.00989","author":"Zhou Jiaxu","year":"2024","unstructured":"Jen-tse Huang, Jiaxu Zhou, Tailin Jin, Xuhui Zhou, Zixi Chen, Wenxuan Wang, Youliang Yuan, Michael R Lyu, and Maarten Sap. 2024. On the resilience of llm-based multi-agent collaboration with faulty agents. arXiv preprint arXiv:2408.00989 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Flooding spread of manipulated knowledge in llm-based multi-agent communities. arXiv preprint arXiv:2407.07791","author":"Ju Tianjie","year":"2024","unstructured":"Tianjie Ju, Yiting Wang, Xinbei Ma, Pengzhou Cheng, Haodong Zhao, Yulong Wang, Lifeng Liu, Jian Xie, Zhuosheng Zhang, and Gongshen Liu. 2024. Flooding spread of manipulated knowledge in llm-based multi-agent communities. arXiv preprint arXiv:2407.07791 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Prompt infection: Llm-to-llm prompt injection within multi-agent systems. arXiv preprint arXiv:2410.07283","author":"Lee Donghyun","year":"2024","unstructured":"Donghyun Lee and Mo Tiwari. 2024. Prompt infection: Llm-to-llm prompt injection within multi-agent systems. arXiv preprint arXiv:2410.07283 (2024)."},{"key":"e_1_3_2_1_19_1","first-page":"51991","article-title":"Camel: Communicative agents for'' mind'' exploration of large language model society","volume":"36","author":"Li Guohao","year":"2023","unstructured":"Guohao Li, Hasan Hammoud, Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem. 2023. Camel: Communicative agents for'' mind'' exploration of large language model society. Advances in Neural Information Processing Systems, Vol. 36 (2023), 51991-52008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","volume-title":"Encouraging divergent thinking in large language models through multi-agent debate. arXiv preprint arXiv:2305.19118","author":"Liang Tian","year":"2023","unstructured":"Tian Liang, Zhiwei He, Wenxiang Jiao, Xing Wang, Yan Wang, Rui Wang, Yujiu Yang, Shuming Shi, and Zhaopeng Tu. 2023. Encouraging divergent thinking in large language models through multi-agent debate. arXiv preprint arXiv:2305.19118 (2023)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.992"},{"key":"e_1_3_2_1_22_1","volume-title":"Truthfulqa: Measuring how models mimic human falsehoods. arXiv preprint arXiv:2109.07958","author":"Lin Stephanie","year":"2021","unstructured":"Stephanie Lin, Jacob Hilton, and Owain Evans. 2021. Truthfulqa: Measuring how models mimic human falsehoods. arXiv preprint arXiv:2109.07958 (2021)."},{"key":"e_1_3_2_1_23_1","volume-title":"Agentsafe: Safeguarding large language model-based multi-agent systems via hierarchical data management. arXiv preprint arXiv:2503.04392","author":"Mao Junyuan","year":"2025","unstructured":"Junyuan Mao, Fanci Meng, Yifan Duan, Miao Yu, Xiaojun Jia, Junfeng Fang, Yuxuan Liang, Kun Wang, and Qingsong Wen. 2025. Agentsafe: Safeguarding large language model-based multi-agent systems via hierarchical data management. arXiv preprint arXiv:2503.04392 (2025)."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the Conference on Health, Inference, and Learning (Proceedings of Machine Learning Research","volume":"260","author":"Pal Ankit","year":"2022","unstructured":"Ankit Pal, Logesh Kumar Umapathi, and Malaikannan Sankarasubbu. 2022. MedMCQA: A Large-scale Multi-Subject Multi-Choice Dataset for Medical domain Question Answering. In Proceedings of the Conference on Health, Inference, and Learning (Proceedings of Machine Learning Research, Vol. 174), Gerardo Flores, George H Chen, Tom Pollard, Joyce C Ho, and Tristan Naumann (Eds.). PMLR, 248-260. https:\/\/proceedings.mlr.press\/v174\/pal22a.html"},{"key":"e_1_3_2_1_25_1","volume-title":"Commonsenseqa: A question answering challenge targeting commonsense knowledge. arXiv preprint arXiv:1811.00937","author":"Talmor Alon","year":"2018","unstructured":"Alon Talmor, Jonathan Herzig, Nicholas Lourie, and Jonathan Berant. 2018. Commonsenseqa: A question answering challenge targeting commonsense knowledge. arXiv preprint arXiv:1811.00937 (2018)."},{"key":"e_1_3_2_1_26_1","unstructured":"Qwen Team. 2024. Qwen2.5: A Party of Foundation Models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"e_1_3_2_1_27_1","volume-title":"Multi-agent systems execute arbitrary malicious code. arXiv preprint arXiv:2503.12188","author":"Triedman Harold","year":"2025","unstructured":"Harold Triedman, Rishi Jha, and Vitaly Shmatikov. 2025. Multi-agent systems execute arbitrary malicious code. arXiv preprint arXiv:2503.12188 (2025)."},{"key":"e_1_3_2_1_28_1","volume-title":"G-safeguard: A topology-guided security lens and treatment on llm-based multi-agent systems. arXiv preprint arXiv:2502.11127","author":"Wang Shilong","year":"2025","unstructured":"Shilong Wang, Guibin Zhang, Miao Yu, Guancheng Wan, Fanci Meng, Chongye Guo, Kun Wang, and Yang Wang. 2025. G-safeguard: A topology-guided security lens and treatment on llm-based multi-agent systems. arXiv preprint arXiv:2502.11127 (2025)."},{"key":"e_1_3_2_1_29_1","volume-title":"First Conference on Language Modeling.","author":"Wu Qingyun","year":"2024","unstructured":"Qingyun Wu, Gagan Bansal, Jieyu Zhang, Yiran Wu, Beibin Li, Erkang Zhu, Li Jiang, Xiaoyun Zhang, Shaokun Zhang, Jiale Liu, et al., 2024. Autogen: Enabling next-gen LLM applications via multi-agent conversations. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_30_1","volume-title":"Netsafe: Exploring the topological safety of multi-agent networks. arXiv preprint arXiv:2410.15686","author":"Yu Miao","year":"2024","unstructured":"Miao Yu, Shilong Wang, Guibin Zhang, Junyuan Mao, Chenlong Yin, Qijiong Liu, Qingsong Wen, Kun Wang, and Yang Wang. 2024. Netsafe: Exploring the topological safety of multi-agent networks. arXiv preprint arXiv:2410.15686 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"Psysafe: A comprehensive framework for psychological-based attack, defense, and evaluation of multi-agent system safety. arXiv preprint arXiv:2401.11880","author":"Zhang Zaibin","year":"2024","unstructured":"Zaibin Zhang, Yongting Zhang, Lijun Li, Hongzhi Gao, Lijun Wang, Huchuan Lu, Feng Zhao, Yu Qiao, and Jing Shao. 2024. Psysafe: A comprehensive framework for psychological-based attack, defense, and evaluation of multi-agent system safety. arXiv preprint arXiv:2401.11880 (2024)."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792462","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:47:59Z","timestamp":1783151279000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792462"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":31,"alternative-id":["10.1145\/3774904.3792462","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792462","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}