{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:25:51Z","timestamp":1776882351240,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"name":"National Science Foundation for Distinguished Young Scholars","award":["62125604"],"award-info":[{"award-number":["62125604"]}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20240826"],"award-info":[{"award-number":["GZC20240826"]}]},{"name":"China Postdoctoral Science Foundation","award":["2024M761679"],"award-info":[{"award-number":["2024M761679"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754561","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:38:54Z","timestamp":1761377934000},"page":"11756-11765","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["JPS: Jailbreak Multimodal Large Language Models with Collaborative Visual Perturbation and Textual Steering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-1213-0916","authenticated-orcid":false,"given":"Renmiao","family":"Chen","sequence":"first","affiliation":[{"name":"CoAI, DCST, Tsinghua University, Beijing, China and Zhipu AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2333-1064","authenticated-orcid":false,"given":"Shiyao","family":"Cui","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4557-7546","authenticated-orcid":false,"given":"Xuancheng","family":"Huang","sequence":"additional","affiliation":[{"name":"Zhipu.AI, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0497-7903","authenticated-orcid":false,"given":"Chengwei","family":"Pan","sequence":"additional","affiliation":[{"name":"Beihang Uinveristy, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0614-5699","authenticated-orcid":false,"given":"Victor Shea-Jay","family":"Huang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5270-3801","authenticated-orcid":false,"given":"QingLin","family":"Zhang","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4115-7340","authenticated-orcid":false,"given":"Xuan","family":"Ouyang","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9601-3991","authenticated-orcid":false,"given":"Zhexin","family":"Zhang","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6524-9195","authenticated-orcid":false,"given":"Hongning","family":"Wang","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7111-1849","authenticated-orcid":false,"given":"Minlie","family":"Huang","sequence":"additional","affiliation":[{"name":"CoAI group, DCST, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2310.08419"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2024.3456150"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2311.05608"},{"key":"e_1_3_2_2_5_1","volume-title":"European Conference on Computer Vision. Springer, 388-404","author":"Gou Yunhao","year":"2024","unstructured":"Yunhao Gou, Kai Chen, Zhili Liu, Lanqing Hong, Hang Xu, Zhenguo Li, Dit-Yan Yeung, James T Kwok, and Yu Zhang. 2024. Eyes closed, safety on: Protecting multimodal llms via image-to-text transformation. In European Conference on Computer Vision. Springer, 388-404."},{"key":"e_1_3_2_2_6_1","volume-title":"VLSBench: Unveiling Visual Leakage in Multimodal Safety. CoRR","author":"Hu Xuhao","year":"1993","unstructured":"Xuhao Hu, Dongrui Liu, Hao Li, Xuanjing Huang, and Jing Shao. 2024. VLSBench: Unveiling Visual Leakage in Multimodal Safety. CoRR, Vol. abs\/2411.19939 (2024)."},{"key":"e_1_3_2_2_7_1","volume-title":"TIDE: Temporal-Aware Sparse Autoencoders for Interpretable Diffusion Transformers in Image Generation. arXiv preprint arXiv:2503.07050","author":"Shea-Jay Huang Victor","year":"2025","unstructured":"Victor Shea-Jay Huang, Le Zhuo, Yi Xin, Zhaokai Wang, Peng Gao, and Hongsheng Li. 2025. TIDE: Temporal-Aware Sparse Autoencoders for Interpretable Diffusion Transformers in Image Generation. arXiv preprint arXiv:2503.07050 (2025)."},{"key":"e_1_3_2_2_8_1","volume-title":"Medical mllm is vulnerable: Cross-modality jailbreak and mismatched attacks on medical multimodal large language models. arXiv preprint arXiv:2405.20775","author":"Huang Xijie","year":"2024","unstructured":"Xijie Huang, Xinyuan Wang, Hantao Zhang, Yinghao Zhu, Jiawen Xi, Jingkun An, Hao Wang, Hao Liang, and Chengwei Pan. 2024. Medical mllm is vulnerable: Cross-modality jailbreak and mismatched attacks on medical multimodal large language models. arXiv preprint arXiv:2405.20775 (2024)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2411.17491"},{"key":"e_1_3_2_2_10_1","first-page":"34892","volume-title":"Levine (Eds.)","volume":"36","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual Instruction Tuning. In Advances in Neural Information Processing Systems, A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Eds.), Vol. 36. Curran Associates, Inc., 34892-34916. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/6dcf277ea32ce3288914faf369fe6de0-Paper-Conference.pdf"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2411.09259"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72992-8_22"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681379"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01480"},{"key":"e_1_3_2_2_15_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Luo Haochen","year":"2024","unstructured":"Haochen Luo, Jindong Gu, Fengyuan Liu, and Philip Torr. 2024a. An Image Is Worth 1000 Lies: Transferability of Adversarial Images across Prompts on Vision-Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_2_16_1","unstructured":"Weidi Luo Siyuan Ma Xiaogeng Liu Xiaoyu Guo and Chaowei Xiao. 2024b. JailBreakV-28K: A Benchmark for Assessing the Robustness of MultiModal Large Language Models against Jailbreak Attacks. arXiv:2404.03027 [cs.CR]"},{"key":"e_1_3_2_2_17_1","volume-title":"Visual-RolePlay: Universal Jailbreak Attack on MultiModal Large Language Models via Role-playing Image Characte. ArXiv","author":"Ma Siyuan","year":"2077","unstructured":"Siyuan Ma, Weidi Luo, Yu Wang, Xiaogeng Liu, Muhao Chen, Bo Li, and Chaowei Xiao. 2024. Visual-RolePlay: Universal Jailbreak Attack on MultiModal Large Language Models via Role-playing Image Characte. ArXiv, Vol. abs\/2405.20773 (2024). https:\/\/api.semanticscholar.org\/CorpusID:270199716"},{"key":"e_1_3_2_2_18_1","unstructured":"Mantas Mazeika Long Phan Xuwang Yin Andy Zou Zifan Wang Norman Mu Elham Sakhaee Nathaniel Li Steven Basart Bo Li David Forsyth and Dan Hendrycks. 2024. HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal. (2024). arXiv:2402.04249 [cs.LG]"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2402.02309"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i19.30150"},{"key":"e_1_3_2_2_21_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Shayegani Erfan","year":"2024","unstructured":"Erfan Shayegani, Yue Dong, and Nael B. Abu-Ghazaleh. 2024. Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=plmBsXHxgR"},{"key":"e_1_3_2_2_22_1","volume-title":"Iterative Self-Tuning LLMs for Enhanced Jailbreaking Capabilities. arXiv preprint arXiv:2410.18469","author":"Sun Chung-En","year":"2024","unstructured":"Chung-En Sun, Xiaodong Liu, Weiwei Yang, Tsui-Wei Weng, Hao Cheng, Aidan San, Michel Galley, and Jianfeng Gao. 2024. Iterative Self-Tuning LLMs for Enhanced Jailbreaking Capabilities. arXiv preprint arXiv:2410.18469 (2024)."},{"key":"e_1_3_2_2_23_1","unstructured":"Alpha VLLM Team. 2025a. Lumina-mGPT 2.0: Stand-alone Autoregressive Image Modeling. https:\/\/github.com\/Alpha-VLLM\/Lumina-mGPT-2.0"},{"key":"e_1_3_2_2_24_1","unstructured":"OpenGVLab Team. 2024a. InternVL2: Better than the Best-Expanding Performance Boundaries of Open-Source Multimodal Models with the Progressive Scaling Strategy. (2024). https:\/\/internvl.github.io\/blog\/2024-07-02-InternVL-2.0\/"},{"key":"e_1_3_2_2_25_1","unstructured":"Qwen Team. 2024b. Qwen2.5: A Party of Foundation Models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"e_1_3_2_2_26_1","unstructured":"Qwen Team. 2025b. QwQ-32B: Embracing the Power of Reinforcement Learning. https:\/\/qwenlm.github.io\/blog\/qwq-32b\/"},{"key":"e_1_3_2_2_27_1","volume-title":"Heuristic-Induced Multimodal Risk Distribution Jailbreak Attack for Multimodal Large Language Models. CoRR","author":"Teng Ma","year":"2024","unstructured":"Ma Teng, Xiaojun Jia, Ranjie Duan, Li Xinfeng, Yihao Huang, Chu Zhixuan, Yang Liu, and Wenqi Ren. 2024. Heuristic-Induced Multimodal Risk Distribution Jailbreak Attack for Multimodal Large Language Models. CoRR, Vol. abs\/2412.05934 (2024)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2409.12191"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681092"},{"key":"e_1_3_2_2_30_1","volume-title":"2024 e. IDEATOR: Jailbreaking Large Vision-Language Models Using Themselves. ArXiv","author":"Wang Ruofan","year":"2024","unstructured":"Ruofan Wang, Bo Wang, Xiaosen Wang, Xingjun Ma, and Yu-Gang Jiang. 2024 e. IDEATOR: Jailbreaking Large Vision-Language Models Using Themselves. ArXiv, Vol. abs\/2411.00827 (2024). https:\/\/api.semanticscholar.org\/CorpusID:273811948"},{"key":"e_1_3_2_2_31_1","volume-title":"2024 f. Cross-Modality Safety Alignment. CoRR","author":"Wang Siyin","year":"2024","unstructured":"Siyin Wang, Xingsong Ye, Qinyuan Cheng, Junwen Duan, Shimin Li, Jinlan Fu, Xipeng Qiu, and Xuanjing Huang. 2024 f. Cross-Modality Safety Alignment. CoRR, Vol. abs\/2406.15279 (2024)."},{"key":"e_1_3_2_2_32_1","volume-title":"Renmiao Chen, Hao Wang, Chengwei Pan, Lei Sha, and Minlie Huang.","author":"Wang Xinyuan","year":"2024","unstructured":"Xinyuan Wang, Victor Shea-Jay Huang, Renmiao Chen, Hao Wang, Chengwei Pan, Lei Sha, and Minlie Huang. 2024b. BlackDAN: A Black-Box Multi-Objective Approach for Effective and Contextual Jailbreaking of Large Language Models. arXiv preprint arXiv:2410.09804 (2024)."},{"key":"e_1_3_2_2_33_1","volume-title":"European Conference on Computer Vision. Springer, 77-94","author":"Wang Yu","year":"2024","unstructured":"Yu Wang, Xiaogeng Liu, Yu Li, Muhao Chen, and Chaowei Xiao. 2024c. Adashield: Safeguarding multimodal large language models from structure-based attack via adaptive shield prompting. In European Conference on Computer Vision. Springer, 77-94."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2412.00473"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2407.16686"},{"key":"e_1_3_2_2_36_1","volume-title":"Cognitive Overload: Jailbreaking Large Language Models with Overloaded Logical Thinking. In Findings of the Association for Computational Linguistics: NAACL 2024","author":"Xu Nan","year":"2024","unstructured":"Nan Xu, Fei Wang, Ben Zhou, Bangzheng Li, Chaowei Xiao, and Muhao Chen. 2024. Cognitive Overload: Jailbreaking Large Language Models with Overloaded Logical Thinking. In Findings of the Association for Computational Linguistics: NAACL 2024, Mexico City, Mexico, June 16-21, 2024, Kevin Duh, Helena G\u00f3mez-Adorno, and Steven Bethard (Eds.). Association for Computational Linguistics, 3526-3548."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2406.04031"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.773"},{"key":"e_1_3_2_2_39_1","unstructured":"Zhexin Zhang Leqi Lei Junxiao Yang Xijie Huang Yida Lu Shiyao Cui Renmiao Chen Qinglin Zhang Xinyuan Wang Hao Wang et al. 2025. AISafetyLab: A Comprehensive Framework for AI Safety Evaluation and Improvement. arXiv preprint arXiv:2502.16776 (2025)."},{"key":"e_1_3_2_2_40_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2307.15043"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2407.02534"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754561","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:10:08Z","timestamp":1765339808000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754561"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":42,"alternative-id":["10.1145\/3746027.3754561","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754561","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}