{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:11:19Z","timestamp":1765357879922,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755236","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:38Z","timestamp":1761377198000},"page":"11414-11423","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Secure Tug-of-War (SecTOW): Iterative Defense-Attack Training with Reinforcement Learning for Multimodal Model Security"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6070-8470","authenticated-orcid":false,"given":"Muzhi","family":"Dai","sequence":"first","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6458-5266","authenticated-orcid":false,"given":"Shixuan","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7666-4020","authenticated-orcid":false,"given":"Zhiyuan","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6000-8168","authenticated-orcid":false,"given":"Junyu","family":"Gao","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Beijing, China and Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7917-1628","authenticated-orcid":false,"given":"Hao","family":"Sun","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0019-4197","authenticated-orcid":false,"given":"Xuelong","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence (TeleAI), China Telecom, China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.662"},{"key":"e_1_3_2_1_2_1","volume-title":"Abusing images and sounds for indirect instruction injection in multi-modal LLMs. arXiv preprint arXiv:2307.10490","author":"Bagdasaryan Eugene","year":"2023","unstructured":"Eugene Bagdasaryan, Tsung-Yin Hsieh, Ben Nassi, and Vitaly Shmatikov. 2023. Abusing images and sounds for indirect instruction injection in multi-modal LLMs. arXiv preprint arXiv:2307.10490 (2023)."},{"key":"e_1_3_2_1_3_1","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding Localization Text Reading and Beyond. arXiv:2308.12966 [cs.CV] https:\/\/arxiv.org\/abs\/2308.12966"},{"key":"e_1_3_2_1_4_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)."},{"key":"e_1_3_2_1_5_1","volume-title":"Image hijacks: Adversarial images can control generative models at runtime. arXiv preprint arXiv:2309.00236","author":"Bailey Luke","year":"2023","unstructured":"Luke Bailey, Euan Ong, Stuart Russell, and Scott Emmons. 2023. Image hijacks: Adversarial images can control generative models at runtime. arXiv preprint arXiv:2309.00236 (2023)."},{"key":"e_1_3_2_1_6_1","first-page":"61478","article-title":"Are aligned neural networks adversarially aligned","volume":"36","author":"Carlini Nicholas","year":"2023","unstructured":"Nicholas Carlini, Milad Nasr, Christopher A Choquette-Choo, Matthew Jagielski, Irena Gao, Pang Wei W Koh, Daphne Ippolito, Florian Tramer, and Ludwig Schmidt. 2023. Are aligned neural networks adversarially aligned? Advances in Neural Information Processing Systems, Vol. 36 (2023), 61478-61500.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","volume-title":"Kartikeya Upasani, and Mahesh Pasupuleti.","author":"Chi Jianfeng","year":"2024","unstructured":"Jianfeng Chi, Ujjwal Karn, Hongyuan Zhan, Eric Smith, Javier Rando, Yiming Zhang, Kate Plawiak, Zacharie Delpierre Coudert, Kartikeya Upasani, and Mahesh Pasupuleti. 2024. Llama guard 3 vision: Safeguarding human-ai image understanding conversations. arXiv preprint arXiv:2411.10414 (2024)."},{"key":"e_1_3_2_1_8_1","unstructured":"Muzhi Dai Shixuan Liu and Qingyi Si. 2025a. Stable Reinforcement Learning for Efficient Reasoning. arXiv:2505.18086 [cs.AI] https:\/\/arxiv.org\/abs\/2505.18086"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Muzhi Dai Jiashuo Sun Zhiyuan Zhao Shixuan Liu Rui Li Junyu Gao and Xuelong Li. 2025b. From Captions to Rewards (CAREVL): Leveraging Large Language Model Experts for Enhanced Reward Modeling in Large Vision-Language Models. arXiv:2503.06260 [cs.CV] https:\/\/arxiv.org\/abs\/2503.06260","DOI":"10.1145\/3746027.3755697"},{"key":"e_1_3_2_1_10_1","unstructured":"Muzhi Dai Chenxu Yang and Qingyi Si. 2025c. S-GRPO: Early Exit via Reinforcement Learning in Reasoning Models. arXiv:2505.07686 [cs.AI] https:\/\/arxiv.org\/abs\/2505.07686"},{"key":"e_1_3_2_1_11_1","volume-title":"Rethinking Bottlenecks in Safety Fine-Tuning of Vision Language Models. arXiv preprint arXiv:2501.18533","author":"Ding Yi","year":"2025","unstructured":"Yi Ding, Lijun Li, Bing Cao, and Jing Shao. 2025. Rethinking Bottlenecks in Safety Fine-Tuning of Vision Language Models. arXiv preprint arXiv:2501.18533 (2025)."},{"key":"e_1_3_2_1_12_1","unstructured":"Hanze Dong Wei Xiong Deepanshu Goyal Yihan Zhang Winnie Chow Rui Pan Shizhe Diao Jipeng Zhang SHUM KaShun and Tong Zhang. [n.d.]. RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment. Transactions on Machine Learning Research ([n.d.])."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.26599\/TST.2019.9010029"},{"key":"e_1_3_2_1_14_1","volume-title":"Figstep: Jailbreaking large vision-language models via typographic visual prompts. arXiv preprint arXiv:2311.05608","author":"Gong Yichen","year":"2023","unstructured":"Yichen Gong, Delong Ran, Jinyuan Liu, Conglei Wang, Tianshuo Cong, Anyu Wang, Sisi Duan, and Xiaoyun Wang. 2023. Figstep: Jailbreaking large vision-language models via typographic visual prompts. arXiv preprint arXiv:2311.05608 (2023)."},{"key":"e_1_3_2_1_15_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_16_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_17_1","volume-title":"The vllm safety paradox: Dual ease in jailbreak attack and defense. arXiv preprint arXiv:2411.08410","author":"Guo Yangyang","year":"2024","unstructured":"Yangyang Guo, Fangkai Jiao, Liqiang Nie, and Mohan Kankanhalli. 2024. The vllm safety paradox: Dual ease in jailbreak attack and defense. arXiv preprint arXiv:2411.08410 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2017.09.056"},{"key":"e_1_3_2_1_19_1","volume-title":"Break the breakout: Reinventing lm defense against jailbreak attacks with self-refinement. arXiv preprint arXiv:2402.15180","author":"Kim Heegyu","year":"2024","unstructured":"Heegyu Kim, Sehyun Yuk, and Hyunsouk Cho. 2024. Break the breakout: Reinventing lm defense against jailbreak attacks with self-refinement. arXiv preprint arXiv:2402.15180 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1138"},{"key":"e_1_3_2_1_21_1","unstructured":"Lei Li Yuwei Yin Shicheng Li Liang Chen Peiyi Wang Shuhuai Ren Mukai Li Yazheng Yang Jingjing Xu Xu Sun et al. 2023. A Large-Scale Dataset towards Multi-Modal Multilingual Instruction Tuning. arXiv preprint arXiv:2306.04387 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Internal Activation Revision: Safeguarding Vision Language Models Without Parameter Update. arXiv preprint arXiv:2501.16378","author":"Li Qing","year":"2025","unstructured":"Qing Li, Jiahui Geng, Zongxiong Chen, Kun Song, Lei Ma, and Fakhri Karray. 2025. Internal Activation Revision: Safeguarding Vision Language Models Without Parameter Update. arXiv preprint arXiv:2501.16378 (2025)."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. 29128-29163","author":"Li Ziniu","year":"2024","unstructured":"Ziniu Li, Tian Xu, Yushun Zhang, Zhihang Lin, Yang Yu, Ruoyu Sun, and Zhi-Quan Luo. 2024. ReMax: a simple, effective, and efficient reinforcement learning method for aligning large language models. In Proceedings of the 41st International Conference on Machine Learning. 29128-29163."},{"key":"e_1_3_2_1_24_1","volume-title":"A survey of attacks on large vision-language models: Resources, advances, and future trends. arXiv preprint arXiv:2407.07403","author":"Liu Daizong","year":"2024","unstructured":"Daizong Liu, Mingyu Yang, Xiaoye Qu, Pan Zhou, Yu Cheng, and Wei Hu. 2024a. A survey of attacks on large vision-language models: Resources, advances, and future trends. arXiv preprint arXiv:2407.07403 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"European Conference on Computer Vision. Springer, 386-403","author":"Liu Xin","year":"2024","unstructured":"Xin Liu, Yichen Zhu, Jindong Gu, Yunshi Lan, Chao Yang, and Yu Qiao. 2024b. Mm-safetybench: A benchmark for safety evaluation of multimodal large language models. In European Conference on Computer Vision. Springer, 386-403."},{"key":"e_1_3_2_1_26_1","volume-title":"Jailbreakv-28k: A benchmark for assessing the robustness of multimodal large language models against jailbreak attacks. arXiv e-prints","author":"Luo Weidi","year":"2024","unstructured":"Weidi Luo, Siyuan Ma, Xiaogeng Liu, Xiaoyu Guo, and Chaowei Xiao. 2024. Jailbreakv-28k: A benchmark for assessing the robustness of multimodal large language models against jailbreak attacks. arXiv e-prints (2024), arXiv-2404."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1153"},{"key":"e_1_3_2_1_28_1","unstructured":"OpenAI: Aaron Hurst Adam Lerer Adam P. Goucher Adam Perelman Aditya Ramesh et al. 2024a. GPT-4o System Card. arXiv:2410.21276 [cs.CL] https:\/\/arxiv.org\/abs\/2410.21276"},{"key":"e_1_3_2_1_29_1","unstructured":"OpenAI Josh Achiam Steven Adler Sandhini Agarwal et al. 2024b. GPT-4 Technical Report. arXiv:2303.08774 [cs.CL] https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.895"},{"key":"e_1_3_2_1_31_1","volume-title":"Visual adversarial examples jailbreak large language models. CoRR","author":"Qi Xiangyu","year":"2023","unstructured":"Xiangyu Qi, Kaixuan Huang, Ashwinee Panda, Mengdi Wang, and Prateek Mittal. 2023. Visual adversarial examples jailbreak large language models. CoRR (2023)."},{"key":"e_1_3_2_1_32_1","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume":"36","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems, Vol. 36 (2023), 53728-53741.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_33_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_34_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zhang, Mingchuan Zhang, YK Li, Y Wu, et al., 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40846-022-00696-6"},{"key":"e_1_3_2_1_36_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. CoRR","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024a. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. CoRR, Vol. abs\/2409.12191 (2024)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"e_1_3_2_1_38_1","volume-title":"European Conference on Computer Vision. Springer, 77-94","author":"Wang Yu","year":"2024","unstructured":"Yu Wang, Xiaogeng Liu, Yu Li, Muhao Chen, and Chaowei Xiao. 2024b. Adashield: Safeguarding multimodal large language models from structure-based attack via adaptive shield prompting. In European Conference on Computer Vision. Springer, 77-94."},{"key":"e_1_3_2_1_39_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_40_1","volume-title":"Safebench: A safety evaluation framework for multimodal large language models. arXiv preprint arXiv:2410.18927","author":"Ying Zonghao","year":"2024","unstructured":"Zonghao Ying, Aishan Liu, Siyuan Liang, Lei Huang, Jinyang Guo, Wenbo Zhou, Xianglong Liu, and Dacheng Tao. 2024. Safebench: A safety evaluation framework for multimodal large language models. arXiv preprint arXiv:2410.18927 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"Dapo: An open-source llm reinforcement learning system at scale. arXiv preprint arXiv:2503.14476","author":"Yu Qiying","year":"2025","unstructured":"Qiying Yu, Zheng Zhang, Ruofei Zhu, Yufeng Yuan, Xiaochen Zuo, Yu Yue, Weinan Dai, Tiantian Fan, Gaohong Liu, Lingjun Liu, et al., 2025. Dapo: An open-source llm reinforcement learning system at scale. arXiv preprint arXiv:2503.14476 (2025)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01310"},{"key":"e_1_3_2_1_43_1","volume-title":"Rrhf: Rank responses to align language models with human feedback without tears. arXiv preprint arXiv:2304.05302","author":"Yuan Zheng","year":"2023","unstructured":"Zheng Yuan, Hongyi Yuan, Chuanqi Tan, Wei Wang, Songfang Huang, and Fei Huang. 2023. Rrhf: Rank responses to align language models with human feedback without tears. arXiv preprint arXiv:2304.05302 (2023)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"e_1_3_2_1_45_1","volume-title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark. arXiv preprint arXiv:2409.02813","author":"Yue Xiang","year":"2024","unstructured":"Xiang Yue, Tianyu Zheng, Yuansheng Ni, Yubo Wang, Kai Zhang, Shengbang Tong, Yuxuan Sun, Botao Yu, Ge Zhang, Huan Sun, Yu Su, Wenhu Chen, and Graham Neubig. 2024b. MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark. arXiv preprint arXiv:2409.02813 (2024)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAES.2024.3404360"},{"key":"e_1_3_2_1_47_1","volume-title":"Safety fine-tuning at (almost) no cost: A baseline for vision large language models. arXiv preprint arXiv:2402.02207","author":"Zong Yongshuo","year":"2024","unstructured":"Yongshuo Zong, Ondrej Bohdal, Tingyang Yu, Yongxin Yang, and Timothy Hospedales. 2024. Safety fine-tuning at (almost) no cost: A baseline for vision large language models. arXiv preprint arXiv:2402.02207 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755236","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:50:13Z","timestamp":1765309813000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755236"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":47,"alternative-id":["10.1145\/3746027.3755236","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755236","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}