{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:26:34Z","timestamp":1765308394381,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","funder":[{"DOI":"10.13039\/100017052","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 62306320, No. 61976217"],"award-info":[{"award-number":["No. 62306320, No. 61976217"]}],"id":[{"id":"10.13039\/100017052","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Open Project Program of State Key Lab. for Novel Software Technology","award":["No. KFKT2024B32"],"award-info":[{"award-number":["No. KFKT2024B32"]}]},{"DOI":"10.13039\/501100004608","name":"Natural Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["No. BK20231063"],"award-info":[{"award-number":["No. BK20231063"]}],"id":[{"id":"10.13039\/501100004608","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755068","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"3380-3389","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Reversible Privacy Preserving on Vision-Language Models via Adversarial Multimodal Key"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8007-5583","authenticated-orcid":false,"given":"Peng","family":"Ying","sequence":"first","affiliation":[{"name":"School of Compute Science and Technology \/ School of Artificial Intelligence, China University of Mining and Technology, Xuzhou, Jiangsu, China and Mine Digitization Engineering Research Center of the Ministry of Education, Xuzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3364-8703","authenticated-orcid":false,"given":"Zhongnian","family":"Li","sequence":"additional","affiliation":[{"name":"School of Compute Science and Technology \/ School of Artificial Intelligence, China University of Mining and Technology, Xuzhou, Jiangsu, China and Mine Digitization Engineering Research Center of the Ministry of Education, Xuzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3836-6487","authenticated-orcid":false,"given":"Meng","family":"Wei","sequence":"additional","affiliation":[{"name":"School of Compute Science and Technology \/ School of Artificial Intelligence, China University of Mining and Technology, Xuzhou, Jiangsu, China and Mine Digitization Engineering Research Center of the Ministry of Education, Xuzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6973-799X","authenticated-orcid":false,"given":"Xinzheng","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Compute Science and Technology \/ School of Artificial Intelligence, China University of Mining and Technology, Xuzhou, Jiangsu, China, State Key Lab. for Novel Software Technology, Nanjing University, Nanjing, Jiangsu, China, and Mine Digitization Engineering Research Center of the Ministry of Education, Xuzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/2976749.2978318"},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022","author":"Alayrac Jean-Baptiste","year":"2022","unstructured":"Jean-Baptiste Alayrac, Jeff Donahue, Pauline Luc, and et al., 2022. Flamingo: a Visual Language Model for Few-Shot Learning. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022."},{"key":"e_1_3_2_1_3_1","volume-title":"Gemini: A Family of Highly Capable Multimodal Models. CoRR","author":"Anil Rohan","year":"2023","unstructured":"Rohan Anil, Sebastian Borgeaud, Yonghui Wu, and et al., 2023. Gemini: A Family of Highly Capable Multimodal Models. CoRR, Vol. abs\/2312.11805 (2023). arXiv:2312.11805"},{"key":"e_1_3_2_1_4_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Bailey Luke","year":"2024","unstructured":"Luke Bailey, Euan Ong, Stuart Russell, and Scott Emmons. 2024. Image Hijacks: Adversarial Images can Control Generative Models at Runtime. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net."},{"key":"e_1_3_2_1_5_1","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020","author":"Brown Tom B.","year":"2020","unstructured":"Tom B. Brown, Benjamin Mann, Nick Ryder, and et al., 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, December 6-12, 2020, virtual."},{"key":"e_1_3_2_1_6_1","unstructured":"Simone Caldarella Massimiliano Mancini Elisa Ricci and et al. 2024. The Phantom Menace: Unmasking Privacy Leakages in Vision-Language Models. CoRR Vol. abs\/2408.01228 (2024). arXiv:2408.01228"},{"key":"e_1_3_2_1_7_1","first-page":"61478","article-title":"Are aligned neural networks adversarially aligned?. In Advances in Neural Information Processing Systems, Vol. 36. Curran Associates","author":"Carlini Nicholas","year":"2023","unstructured":"Nicholas Carlini, Milad Nasr, Choquette-Choo, and et al., 2023. Are aligned neural networks adversarially aligned?. In Advances in Neural Information Processing Systems, Vol. 36. Curran Associates, Inc., 61478-61500.","journal-title":"Inc."},{"key":"e_1_3_2_1_8_1","first-page":"2633","volume-title":"30th USENIX Security Symposium, USENIX Security 2021","author":"Carlini Nicholas","year":"2021","unstructured":"Nicholas Carlini, Florian Tram\u00e8r, Eric Wallace, and et al., 2021. Extracting Training Data from Large Language Models. In 30th USENIX Security Symposium, USENIX Security 2021, August 11-13, 2021. USENIX Association, 2633-2650."},{"key":"e_1_3_2_1_9_1","unstructured":"Wenliang Dai Junnan Li Dongxu Li and et al. 2023. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023 NeurIPS 2023 New Orleans LA USA December 10 - 16 2023."},{"key":"e_1_3_2_1_10_1","unstructured":"Yichen Gong Delong Ran Jinyuan Liu and et al. 2023. FigStep: Jailbreaking Large Vision-language Models via Typographic Visual Prompts. CoRR Vol. abs\/2311.05608 (2023). arXiv:2311.05608"},{"key":"e_1_3_2_1_11_1","unstructured":"Tianle Gu Zeyang Zhou Kexin Huang and et al. 2024. MLLMGuard: A Multi-dimensional Safety Evaluation Suite for Multimodal Large Language Models. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024 NeurIPS 2024 Vancouver BC Canada December 10 - 15 2024."},{"key":"e_1_3_2_1_12_1","volume-title":"UK","volume":"743","author":"Gu Xiuye","year":"2020","unstructured":"Xiuye Gu, Weixin Luo, Michael S. Ryoo, and et al., 2020. Password-Conditioned Anonymization and Deanonymization with Face Identity Transformers. In Computer Vision - ECCV 2020 - 16th European Conference, Glasgow, UK, August 23-28, 2020, Proceedings, Part XXIII (Lecture Notes in Computer Science, Vol. 12368). Springer, 727-743."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3594869"},{"volume-title":"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics","author":"He Xuanli","key":"e_1_3_2_1_14_1","unstructured":"Xuanli He, Lingjuan Lyu, Lichao Sun, and et al., 2021. Model Extraction and Adversarial Transferability, Your BERT is Vulnerable!. In Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics, Online, 2006-2012."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671573"},{"key":"e_1_3_2_1_17_1","unstructured":"Haoran Li Yulin Chen Jinglong Luo and et al. 2023. Privacy in Large Language Models: Attacks Defenses and Future Directions. CoRR Vol. abs\/2310.10383 (2023). arXiv:2310.10383"},{"key":"e_1_3_2_1_18_1","volume-title":"Vl-trojan: Multimodal instruction backdoor attacks against autoregressive visual language models. International Journal of Computer Vision","author":"Liang Jiawei","year":"2025","unstructured":"Jiawei Liang, Siyuan Liang, Aishan Liu, and Xiaochun Cao. 2025. Vl-trojan: Multimodal instruction backdoor attacks against autoregressive visual language models. International Journal of Computer Vision (2025), 1-20."},{"key":"e_1_3_2_1_19_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2023), 34892-34916."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","unstructured":"Haoyu Lu Wen Liu Bo Zhang and et al. 2024. DeepSeek-VL: Towards Real-World Vision-Language Understanding. CoRR Vol. abs\/2403.05525 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2403.05525 arXiv:2403.05525","DOI":"10.48550\/ARXIV.2403.05525"},{"key":"e_1_3_2_1_21_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Luo Haochen","year":"2024","unstructured":"Haochen Luo, Jindong Gu, Fengyuan Liu, and Philip Torr. 2024. An Image Is Worth 1000 Lies: Transferability of Adversarial Images across Prompts on Vision-Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_1_22_1","volume-title":"6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3, 2018, Conference Track Proceedings. OpenReview.net.","author":"Madry Aleksander","year":"2018","unstructured":"Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, and Adrian Vladu. 2018. Towards Deep Learning Models Resistant to Adversarial Attacks. In 6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 - May 3, 2018, Conference Track Proceedings. OpenReview.net."},{"key":"e_1_3_2_1_23_1","volume-title":"Differentially private decoding in large language models. arXiv preprint arXiv:2205.13621","author":"Majmudar Jimit","year":"2022","unstructured":"Jimit Majmudar, Christophe Dupuy, Charith Peris, Sami Smaili, Rahul Gupta, and Richard Zemel. 2022. Differentially private decoding in large language models. arXiv preprint arXiv:2205.13621 (2022)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.719"},{"key":"e_1_3_2_1_25_1","unstructured":"OpenAI. 2022. Introducing ChatGPT. https:\/\/openai.com\/blog\/chatgpt"},{"key":"e_1_3_2_1_26_1","unstructured":"OpenAI. 2023. GPT-4V(ision) System Card. https:\/\/openai.com\/research\/gpt-4v-system-card"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Xiangyu Qi Kaixuan Huang Ashwinee Panda and et al. 2024. Visual Adversarial Examples Jailbreak Aligned Large Language Models. In Thirty-Eighth AAAI Conference on Artificial Intelligence AAAI 2024 Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence IAAI 2024 Fourteenth Symposium on Educational Advances in Artificial Intelligence EAAI 2014 February 20-27 2024 Vancouver Canada. AAAI Press 21527-21536.","DOI":"10.1609\/aaai.v38i19.30150"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021","volume":"8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, and et al., [n.d.]. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event (Proceedings of Machine Learning Research, Vol. 139). PMLR, 8748-8763."},{"key":"e_1_3_2_1_29_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Shayegani Erfan","year":"2024","unstructured":"Erfan Shayegani, Yue Dong, and Nael B. Abu-Ghazaleh. 2024. Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Om Thakkar Swaroop Ramaswamy Rajiv Mathews and et al. 2020. Understanding Unintended Memorization in Federated Learning. CoRR Vol. abs\/2006.07490 (2020). arXiv:2006.07490","DOI":"10.18653\/v1\/2021.privatenlp-1.1"},{"key":"e_1_3_2_1_31_1","volume-title":"Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024","author":"T\u00f6mek\u00e7e Batuhan","year":"2024","unstructured":"Batuhan T\u00f6mek\u00e7e, Mark Vero, Robin Staab, and et al., 2024. Private Attribute Inference from Images with Vision-Language Models. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024."},{"key":"e_1_3_2_1_32_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone and et al. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. CoRR Vol. abs\/2307.09288 (2023). arXiv:2307.09288"},{"key":"e_1_3_2_1_33_1","unstructured":"Peng Wang Shuai Bai Sinan Tan and et al. 2024a. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. CoRR Vol. abs\/2409.12191 (2024)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681092"},{"key":"e_1_3_2_1_35_1","volume-title":"Cross-Modality Safety Alignment. CoRR","author":"Wang Siyin","year":"2024","unstructured":"Siyin Wang, Xingsong Ye, Qinyuan Cheng, Junwen Duan, Shimin Li, Jinlan Fu, Xipeng Qiu, and Xuanjing Huang. 2024c. Cross-Modality Safety Alignment. CoRR, Vol. abs\/2406.15279 (2024)."},{"key":"e_1_3_2_1_36_1","volume-title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, and et al., 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671897"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3344809"},{"key":"e_1_3_2_1_39_1","volume-title":"A Multi-perspective Benchmark on Privacy Assessment for Large Vision-Language Models. CoRR","author":"Zhang Jie","year":"1949","unstructured":"Jie Zhang, Xiangkui Cao, Zhouyu Han, Shiguang Shan, and Xilin Chen. 2024a. Multi-P(^mbox2, )A: A Multi-perspective Benchmark on Privacy Assessment for Large Vision-Language Models. CoRR, Vol. abs\/2412.19496 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Adversarial examples: Opportunities and challenges","author":"Zhang Jiliang","year":"2019","unstructured":"Jiliang Zhang and Chen Li. 2019. Adversarial examples: Opportunities and challenges. IEEE transactions on neural networks and learning systems, Vol. 31, 7 (2019), 2578-2593."},{"key":"e_1_3_2_1_41_1","unstructured":"Yichi Zhang Yao Huang Yitong Sun and et al. 2024b. Benchmarking Trustworthiness of Multimodal Large Language Models: A Comprehensive Study. CoRR Vol. abs\/2406.07057 (2024)."},{"key":"e_1_3_2_1_42_1","first-page":"54111","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Zhao Yunqing","year":"2023","unstructured":"Yunqing Zhao, Tianyu Pang, Chao Du, Xiao Yang, Chongxuan LI, Ngai-Man (Man) Cheung, and Min Lin. 2023. On Evaluating Adversarial Robustness of Large Vision-Language Models. In Advances in Neural Information Processing Systems, Vol. 36. Curran Associates, Inc., 54111-54138."},{"key":"e_1_3_2_1_43_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Junjie Zhu Lin Gu Xiaoxiao Wu and et al. 2023. People Taking Photos That Faces Never Share: Privacy Protection and Fairness Enhancement from Camera to User. In Thirty-Seventh AAAI Conference on Artificial Intelligence AAAI 2023 Thirty-Fifth Conference on Innovative Applications of Artificial Intelligence IAAI 2023 Thirteenth Symposium on Educational Advances in Artificial Intelligence EAAI 2023 Washington DC USA February 7-14 2023. AAAI Press 14646-14654.","DOI":"10.1609\/aaai.v37i12.26712"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755068","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:22:45Z","timestamp":1765308165000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755068"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":44,"alternative-id":["10.1145\/3746027.3755068","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755068","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}