{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,14]],"date-time":"2026-03-14T17:59:20Z","timestamp":1773511160744,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Shaanxi Key Research and Development Program","award":["2022ZDLGY03-04"],"award-info":[{"award-number":["2022ZDLGY03-04"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681184","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"18-27","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["A Unified Understanding of Adversarial Vulnerability Regarding Unimodal Models and Vision-Language Pre-training Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2017-962X","authenticated-orcid":false,"given":"Haonan","family":"Zheng","sequence":"first","affiliation":[{"name":"School of Electronics and Information, Northwestern Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8181-7001","authenticated-orcid":false,"given":"Xinyang","family":"Deng","sequence":"additional","affiliation":[{"name":"School of Electronics and Information, Northwestern Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5429-2748","authenticated-orcid":false,"given":"Wen","family":"Jiang","sequence":"additional","affiliation":[{"name":"School of Electronics and Information, Northwestern Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2393-9016","authenticated-orcid":false,"given":"Wenrui","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Harbin Institute of Technology, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 3674--3683","author":"Anderson Peter","year":"2018","unstructured":"Peter Anderson, Qi Wu, Damien Teney, Jake Bruce, Mark Johnson, Niko S\u00fcnderhauf, Ian Reid, Stephen Gould, and Anton Van Den Hengel. 2018. Vision-andlanguage navigation: interpreting visually-grounded navigation instructions in real environments. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 3674--3683."},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Bao Hangbo","year":"2022","unstructured":"Hangbo Bao, Li Dong, Songhao Piao, and Furu Wei. 2022. BEiT: BERT pretraining of image transformers. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems (NeurIPS), 32897--32912","author":"Bao Hangbo","year":"2022","unstructured":"Hangbo Bao, Wenhui Wang, Li Dong, Qiang Liu, Owais Khan Mohammed, Kriti Aggarwal, Subhojit Som, Songhao Piao, and FuruWei. 2022. Vlmo: unified vision-language pre-training with mixture-of-modality-experts. In Proceedings of the 36th International Conference on Neural Information Processing Systems (NeurIPS), 32897--32912."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.49"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 1597--1607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In Proceedings of the International Conference on Machine Learning (ICML), 1597--1607."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_31"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 2201--2211","author":"Croce Francesco","year":"2021","unstructured":"Francesco Croce and Matthias Hein. 2021. Mind the box: \u03b91-apgd for sparse adversarial attacks on image classifiers. In Proceedings of the International Conference on Machine Learning (ICML), 2201--2211."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00957"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00444"},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Alexey","unstructured":"Alexey Dosovitskiy et al. 2023. An image is worth 16x16 words: transformers for image recognition at scale. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_12_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Goodfellow Ian J","year":"2015","unstructured":"Ian J Goodfellow, Jonathon Shlens, and Christian Szegedy. 2015. Explaining and harnessing adversarial examples. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems (NeurIPS), 21271--21284","author":"Jean-Bastien","unstructured":"Jean-Bastien Grill et al. 2020. Bootstrap your own latent a new approach to self-supervised learning. In Proceedings of the 34th International Conference on Neural Information Processing Systems (NeurIPS), 21271--21284."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 2507--2515","author":"Karmon Danny","year":"2018","unstructured":"Danny Karmon, Daniel Zoran, and Yoav Goldberg. 2018. LaVAN: localized and visible adversarial noise. In Proceedings of the International Conference on Machine Learning (ICML), 2507--2515."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 5583--5594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: vision-and-language transformer without convolution or region supervision. In Proceedings of the International Conference on Machine Learning (ICML), 5583--5594."},{"key":"e_1_3_2_1_18_1","unstructured":"A. Krizhevsky and G. Hinton. 2009. Learning multiple layers of features from tiny images. Handbook of Systemic Autoimmune Diseases."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 12888--12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In Proceedings of the International Conference on Machine Learning (ICML), 12888--12900."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems (NeurIPS), 9694--9705","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: vision and language representation learning with momentum distillation. In Proceedings of the 35th International Conference on Neural Information Processing Systems (NeurIPS), 9694--9705."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.500"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00072"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548138"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611758"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611759"},{"key":"e_1_3_2_1_26_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Lin Jiadong","year":"2020","unstructured":"Jiadong Lin, Chuanbiao Song, Kun He, Liwei Wang, and John E Hopcroft. 2020. Nesterov accelerated gradient and scale invariance for adversarial attacks. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 13th European Conference on Computer Vision (ECCV), 740--755","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: common objects in context. In Proceedings of the 13th European Conference on Computer Vision (ECCV), 740--755."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Fangyu Liu R\u00e9mi Lebret Didier Orel Philippe Sordet and Karl Aberer. 2020. Upgrading the newsroom: an automated image selection system for news articles. ACM Transactions on Multimedia Computing Communications and Applications (TOMM) 1--28.","DOI":"10.1145\/3396520"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00016"},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Madry Aleksander","year":"2018","unstructured":"Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, and Adrian Vladu. 2018. Towards deep learning models resistant to adversarial attacks. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.282"},{"key":"e_1_3_2_1_32_1","article-title":"Mra-net: improving vqa via multi-modal relation attention network","author":"Peng Liang","year":"2020","unstructured":"Liang Peng, Yang Yang, Zheng Wang, Zi Huang, and Heng Tao Shen. 2020. Mra-net: improving vqa via multi-modal relation attention network. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 318--329.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 318--329."},{"key":"e_1_3_2_1_33_1","unstructured":"Zhiliang Peng Li Dong Hangbo Bao Qixiang Ye and Furu Wei. 2022. BEiT v2: masked image modeling with vector-quantized visual tokenizers. In arXiv preprint arXiv:2208.06366."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 8748--8763","author":"Alec","unstructured":"Alec Radford et al. 2021. Learning transferable visual models from natural language supervision. In Proceedings of the International Conference on Machine Learning (ICML), 8748--8763."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1644"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Wenhui","unstructured":"Wenhui Wang et al. 2023. Image as a foreign language: BEiT pretraining for vision and vision-language tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 19175--19186."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00196"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 6629--6638","author":"Wang Xin","year":"2019","unstructured":"Xin Wang, Qiuyuan Huang, Asli Celikyilmaz, Jianfeng Gao, Dinghan Shen, Yuan-Fang Wang, William Yang Wang, and Lei Zhang. 2019. Reinforced crossmodal matching and self-supervised imitation learning for vision-language navigation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 6629--6638."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3249754"},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2730--2739","author":"Xie Cihang","unstructured":"Cihang Xie, Zhishuai Zhang, Yuyin Zhou, Song Bai, Jianyu Wang, Zhou Ren, and Alan L. Yuille. 2019. Improving transferability of adversarial examples with input diversity. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 2730--2739."},{"key":"e_1_3_2_1_43_1","unstructured":"Ning Xie Farley Lai Derek Doran and Asim Kadav. 2019. Visual Entailment: a novel task for fine-grained image understanding. In arXiv preprint:1901.06706."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01522"},{"key":"e_1_3_2_1_45_1","article-title":"Adaptive semi-supervised feature selection for cross-modal retrieval","author":"Yu En","year":"2018","unstructured":"En Yu, Jiande Sun, Jing Li, Xiaojun Chang, Xian-Hua Han, and Alexander G Hauptmann. 2018. Adaptive semi-supervised feature selection for cross-modal retrieval. IEEE Transactions on Multimedia (TMM), 1276--1288.","journal-title":"IEEE Transactions on Multimedia (TMM), 1276--1288."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the 14th European Conference on Computer Vision (ECCV), 69--85","author":"Yu Licheng","unstructured":"Licheng Yu, Patrick Poirson, Shan Yang, Alexander C. Berg, and Tamara L. Berg. 2016. Modeling context in referring expressions. In Proceedings of the 14th European Conference on Computer Vision (ECCV), 69--85."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547801"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612454"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681184","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681184","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:02Z","timestamp":1750295882000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681184"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":48,"alternative-id":["10.1145\/3664647.3681184","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681184","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}