{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:49Z","timestamp":1784138929485,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Guangdong S&amp;T Program","award":["2025B0101130005"],"award-info":[{"award-number":["2025B0101130005"]}]},{"name":"Natural Science Foundation of Guangdong Province of China","award":["2024A1515030166"],"award-info":[{"award-number":["2024A1515030166"]}]},{"name":"Natural Science Foundation of Guangdong Province of China","award":["2025B1515020032"],"award-info":[{"award-number":["2025B1515020032"]}]},{"name":"Innovation Team Project of Guangdong Province","award":["2024KCXTD017"],"award-info":[{"award-number":["2024KCXTD017"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808579","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"3506-3513","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["EVADE-Bench: Multimodal Benchmark for Evaluating and Enhancing Evasive Content Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0624-3340","authenticated-orcid":false,"given":"Ancheng","family":"Xu","sequence":"first","affiliation":[{"name":"Shenzhen Institute of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4501-1826","authenticated-orcid":false,"given":"Zhihao","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4909-9113","authenticated-orcid":false,"given":"Jingpeng","family":"Li","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1559-2996","authenticated-orcid":false,"given":"Guanghu","family":"Yuan","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9177-8152","authenticated-orcid":false,"given":"Longze","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4168-3736","authenticated-orcid":false,"given":"Liang","family":"Yan","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0709-775X","authenticated-orcid":false,"given":"Jiehui","family":"Zhou","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5184-4692","authenticated-orcid":false,"given":"Zhen","family":"Qin","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1744-1513","authenticated-orcid":false,"given":"Hengyu","family":"Chang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3107-3366","authenticated-orcid":false,"given":"Yukun","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2189-9153","authenticated-orcid":false,"given":"Hamid","family":"Alinejad-Rokny","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3814-2728","authenticated-orcid":false,"given":"Min","family":"Yang","sequence":"additional","affiliation":[{"name":"Shenzhen Institutes of Advanced Technology, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Sylvia Worlali Azumah Nelly Elsayed Zag ElSayed Murat Ozer and Amanda La Guardia. 2024. Deep Learning Approaches for Detecting Adversarial Cyberbullying and Hate Speech in Social Networks. arXiv:2406.17793 [cs.LG] https:\/\/arxiv.org\/abs\/2406.17793"},{"key":"e_1_3_2_1_2_1","unstructured":"Zechen Bai Pichao Wang Tianjun Xiao Tong He Zongbo Han Zheng Zhang and Mike Zheng Shou. 2025. Hallucination of Multimodal Large Language Models: A Survey. arXiv:2404.18930 [cs.CV] https:\/\/arxiv.org\/abs\/2404.18930"},{"key":"e_1_3_2_1_3_1","unstructured":"Yukang Chen Fuzhao Xue Dacheng Li Qinghao Hu Ligeng Zhu Xiuyu Li Yunhao Fang Haotian Tang Shang Yang Zhijian Liu and et al. 2024. LongVILA: Scaling Long-Context Visual Language Models for Long Videos. arXiv:2408.10188 [cs.CV] https:\/\/arxiv.org\/abs\/2408.10188"},{"key":"e_1_3_2_1_4_1","volume-title":"Vaibhav Kumar, Faysal Hossain Shezan, Vaibhav Kumar, Vinija Jain, and Aman Chadha.","author":"Chowdhury Arijit Ghosh","year":"2024","unstructured":"Arijit Ghosh Chowdhury, Md Mofijul Islam, Vaibhav Kumar, Faysal Hossain Shezan, Vaibhav Kumar, Vinija Jain, and Aman Chadha. 2024. Breaking Down the Defenses: A Comparative Survey of Attacks on Large Language Models. arXiv:2403.04786 [cs.CR] https:\/\/arxiv.org\/abs\/2403.04786"},{"key":"e_1_3_2_1_5_1","unstructured":"Zixian Guo Ming Liu Qilong Wang Zhilong Ji Jinfeng Bai Lei Zhang and Wangmeng Zuo. 2025. Integrating Visual Interpretation and Linguistic Reasoning for Math Problem Solving. arXiv:2505.17609 [cs.AI] https:\/\/arxiv.org\/abs\/2505.17609"},{"key":"e_1_3_2_1_6_1","unstructured":"William Hackett Lewis Birch Stefan Trawicki Neeraj Suri and Peter Garraghan. 2025. Bypassing Prompt Injection and Jailbreak Detection in LLM Guardrails. arXiv:2504.11168 [cs.CR] https:\/\/arxiv.org\/abs\/2504.11168"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3703155"},{"key":"e_1_3_2_1_8_1","volume-title":"Beyond Utility: Evaluating LLM as Recommender. arXiv:2411.00331 [cs.IR] https:\/\/arxiv.org\/abs\/2411.00331","author":"Jiang Chumeng","year":"2024","unstructured":"Chumeng Jiang, Jiayin Wang, Weizhi Ma, Charles L. A. Clarke, Shuai Wang, Chuhan Wu, and Min Zhang. 2024. Beyond Utility: Evaluating LLM as Recommender. arXiv:2411.00331 [cs.IR] https:\/\/arxiv.org\/abs\/2411.00331"},{"key":"e_1_3_2_1_9_1","unstructured":"Douwe Kiela Hamed Firooz Aravind Mohan Vedanuj Goswami Amanpreet Singh Pratik Ringshia and Davide Testuggine. 2021. The Hateful Memes Challenge: Detecting Hate Speech in Multimodal Memes. arXiv:2005.04790 [cs.AI] https:\/\/arxiv.org\/abs\/2005.04790"},{"key":"e_1_3_2_1_10_1","unstructured":"Chunyuan Li Zhe Gan Zhengyuan Yang Jianwei Yang Linjie Li Lijuan Wang and Jianfeng Gao. 2023. Multimodal Foundation Models: From Specialists to General-Purpose Assistants. arXiv:2309.10020 [cs.CV] https:\/\/arxiv.org\/abs\/2309.10020"},{"key":"e_1_3_2_1_11_1","volume-title":"Yao Yua, Wei An-Hou, Li Ming, Tianyang Wang, Ziqian Bi, and Ming Liu.","author":"Liang Chia Xin","year":"2024","unstructured":"Chia Xin Liang, Pu Tian, Caitlyn Heqi Yin, Yao Yua, Wei An-Hou, Li Ming, Tianyang Wang, Ziqian Bi, and Ming Liu. 2024. A Comprehensive Survey and Guide to Multimodal Large Language Models in Vision-Language Tasks. arXiv:2411.06284 [cs.AI] https:\/\/arxiv.org\/abs\/2411.06284"},{"key":"e_1_3_2_1_12_1","unstructured":"Daizong Liu Mingyu Yang Xiaoye Qu Pan Zhou Yu Cheng and Wei Hu. 2024a. A Survey of Attacks on Large Vision-Language Models: Resources Advances and Future Trends. arXiv:2407.07403 [cs.CV] https:\/\/arxiv.org\/abs\/2407.07403"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Xin Liu Yichen Zhu Jindong Gu Yunshi Lan Chao Yang and Yu Qiao. 2024b. MM-SafetyBench: A Benchmark for Safety Evaluation of Multimodal Large Language Models. arXiv:2311.17600 [cs.CV] https:\/\/arxiv.org\/abs\/2311.17600","DOI":"10.1007\/978-3-031-72992-8_22"},{"key":"e_1_3_2_1_14_1","unstructured":"Renze Lou Kai Zhang and Wenpeng Yin. 2024. Large Language Model Instruction Following: A Survey of Progresses and Challenges. arXiv:2303.10475 [cs.CL] https:\/\/arxiv.org\/abs\/2303.10475"},{"key":"e_1_3_2_1_15_1","unstructured":"Andrea Matarazzo and Riccardo Torlone. 2025. A Survey on Large Language Models with some Insights on their Capabilities and Limitations. arXiv:2501.04040 [cs.CL] https:\/\/arxiv.org\/abs\/2501.04040"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3411990"},{"key":"e_1_3_2_1_17_1","unstructured":"Rudra Murthy Praveen Venkateswaran Prince Kumar and Danish Contractor. 2025. Evaluating the Instruction-following Abilities of Language Models using Knowledge Tasks. arXiv:2410.12972 [cs.CL] https:\/\/arxiv.org\/abs\/2410.12972"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.173"},{"key":"e_1_3_2_1_19_1","unstructured":"OpenAI Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman and et al. 2024. GPT-4 Technical Report. arXiv:2303.08774 [cs.CL] https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_1_20_1","unstructured":"Chester Palen-Michel Ruixiang Wang Yipeng Zhang David Yu Canran Xu and Zhe Wu. 2024. Investigating LLM Applications in E-Commerce. arXiv:2408.12779 [cs.CL] https:\/\/arxiv.org\/abs\/2408.12779"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.739"},{"key":"e_1_3_2_1_22_1","volume-title":"Prism: A Framework for Decoupling and Assessing the Capabilities of VLMs. arXiv:2406.14544 [cs.CV] https:\/\/arxiv.org\/abs\/2406.14544","author":"Qiao Yuxuan","year":"2024","unstructured":"Yuxuan Qiao, Haodong Duan, Xinyu Fang, Junming Yang, Lin Chen, Songyang Zhang, Jiaqi Wang, Dahua Lin, and Kai Chen. 2024. Prism: A Framework for Decoupling and Assessing the Capabilities of VLMs. arXiv:2406.14544 [cs.CV] https:\/\/arxiv.org\/abs\/2406.14544"},{"key":"e_1_3_2_1_23_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, and et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, and et al., 2021. Learning Transferable Visual Models From Natural Language Supervision. arXiv:2103.00020 [cs.CV] https:\/\/arxiv.org\/abs\/2103.00020"},{"key":"e_1_3_2_1_24_1","volume-title":"Syed Raza Bashir, Elham Dolatabadi, Gias Uddin, Christos Emmanouilidis, Rizwan Qureshi, and et al.","author":"Raza Shaina","year":"2025","unstructured":"Shaina Raza, Ashmal Vayani, Aditya Jain, Aravind Narayanan, Vahid Reza Khazaie, Syed Raza Bashir, Elham Dolatabadi, Gias Uddin, Christos Emmanouilidis, Rizwan Qureshi, and et al., 2025. VLDBench: Vision Language Models Disinformation Detection Benchmark. arXiv:2502.11361 [cs.CL] https:\/\/arxiv.org\/abs\/2502.11361"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Qingyang Ren Zilin Jiang Jinghan Cao Sijia Li Chiqu Li Yiyang Liu Shuning Huo Tiange He and Yuan Chen. 2024. A survey on fairness of large language models in e-commerce: progress application and challenge. arXiv:2405.13025 [cs.CL] https:\/\/arxiv.org\/abs\/2405.13025","DOI":"10.18653\/v1\/2024.findings-acl.772"},{"key":"e_1_3_2_1_26_1","volume-title":"Youliang Yuan, Pinjia He, Shuai Wang, and Zhaopeng Tu.","author":"Wang Wenxuan","year":"2025","unstructured":"Wenxuan Wang, Xiaoyuan Liu, Kuiyi Gao, Jen tse Huang, Youliang Yuan, Pinjia He, Shuai Wang, and Zhaopeng Tu. 2025. Can't See the Forest for the Trees: Benchmarking Multimodal Safety Awareness for Multimodal LLMs. arXiv:2502.11184 [cs.CL] https:\/\/arxiv.org\/abs\/2502.11184"},{"key":"e_1_3_2_1_27_1","unstructured":"Xindi Wang Mahsa Salmani Parsa Omidi Xiangyu Ren Mehdi Rezagholizadeh and Armaghan Eshaghi. 2024. Beyond the Limits: A Survey of Techniques to Extend the Context Length in Large Language Models. arXiv:2402.02244 [cs.CL] https:\/\/arxiv.org\/abs\/2402.02244"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1154"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102888"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.6"},{"key":"e_1_3_2_1_31_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv and et al. 2025. Qwen3 Technical Report. arXiv:2505.09388 [cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_1_32_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong and et al. 2025. A Survey of Large Language Models. arXiv:2303.18223 [cs.CL] https:\/\/arxiv.org\/abs\/2303.18223"},{"key":"e_1_3_2_1_33_1","volume-title":"Xuanjing Huang, Yu-Gang Jiang, Nicu Sebe, Dacheng Tao, Luc Van Gool, and Xuming Hu.","author":"Zheng Xu","year":"2025","unstructured":"Xu Zheng, Chenfei Liao, Yuqian Fu, Kaiyu Lei, Yuanhuiyi Lyu, Lutao Jiang, Bin Ren, Jialei Chen, Jiawen Wang, Chengxin Li, Linfeng Zhang, Danda Pani Paudel, Xuanjing Huang, Yu-Gang Jiang, Nicu Sebe, Dacheng Tao, Luc Van Gool, and Xuming Hu. 2025. MLLMs are Deeply Affected by Modality Bias. arXiv:2505.18657 [cs.AI] https:\/\/arxiv.org\/abs\/2505.18657"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.27"},{"key":"e_1_3_2_1_35_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J. Zico Kolter and Matt Fredrikson. 2023. Universal and Transferable Adversarial Attacks on Aligned Language Models. arXiv:2307.15043 [cs.CL] https:\/\/arxiv.org\/abs\/2307.15043"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:24:20Z","timestamp":1784136260000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808579"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":35,"alternative-id":["10.1145\/3805712.3808579","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808579","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}