{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T15:12:52Z","timestamp":1778166772102,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Shaanxi Key Research and Development Program","award":["2022ZDLGY03-04"],"award-info":[{"award-number":["2022ZDLGY03-04"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3686835","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"9749-9758","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Sample-agnostic Adversarial Perturbation for Vision-Language Pre-training Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2017-962X","authenticated-orcid":false,"given":"Haonan","family":"Zheng","sequence":"first","affiliation":[{"name":"School of Electronics and Information, Northwest Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5429-2748","authenticated-orcid":false,"given":"Wen","family":"Jiang","sequence":"additional","affiliation":[{"name":"School of Electronics and Information, Northwestern Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8181-7001","authenticated-orcid":false,"given":"Xinyang","family":"Deng","sequence":"additional","affiliation":[{"name":"School of Electronics and Information, Northwestern Polytechnical University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2393-9016","authenticated-orcid":false,"given":"Wenrui","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Harbin Institute of Technology, Harbin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Bao Hangbo","year":"2022","unstructured":"Hangbo Bao, Li Dong, Songhao Piao, and Furu Wei. 2022. BEiT: BERT pretraining of image transformers. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems (NeurIPS), 32897--32912","author":"Bao Hangbo","year":"2022","unstructured":"Hangbo Bao, Wenhui Wang, Li Dong, Qiang Liu, Owais Khan Mohammed, Kriti Aggarwal, Subhojit Som, Songhao Piao, and FuruWei. 2022. Vlmo: unified vision-language pre-training with mixture-of-modality-experts. In Proceedings of the 36th International Conference on Neural Information Processing Systems (NeurIPS), 32897--32912."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00466"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1646396.1646452"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_7_1","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2018. Bert: pre-training of deep bidirectional transformers for language understanding. In arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_1_8_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Alexey","unstructured":"Alexey Dosovitskiy et al. 2023. An image is worth 16x16 words: transformers for image recognition at scale. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_9_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Goodfellow Ian J","year":"2015","unstructured":"Ian J Goodfellow, Jonathon Shlens, and Christian Szegedy. 2015. Explaining and harnessing adversarial examples. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_10_1","unstructured":"Liang Jiawei Liang Siyuan Luo Man Liu Aishan Han Dongchen Chang Ee-Chien and Cao Xiaochun. 2024. Vl-trojan: multimodal instruction backdoor attacks against autoregressive visual language models. In arXiv preprint arXiv:2402.13851."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 5583--5594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: vision-and-language transformer without convolution or region supervision. In Proceedings of the International Conference on Machine Learning (ICML), 5583--5594."},{"key":"e_1_3_2_1_12_1","unstructured":"A. Krizhevsky and G. Hinton. 2009. Learning multiple layers of features from tiny images. Handbook of Systemic Autoimmune Diseases."},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 12888--12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In Proceedings of the International Conference on Machine Learning (ICML), 12888--12900."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems (NeurIPS), 9694--9705","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: vision and language representation learning with momentum distillation. In Proceedings of the 35th International Conference on Neural Information Processing Systems (NeurIPS), 9694--9705."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611758"},{"key":"e_1_3_2_1_16_1","article-title":"Spiking tucker fusion transformer for audio-visual zero-shot learning","author":"Li Wenrui","year":"2024","unstructured":"Wenrui Li, Penghong Wang, Ruiqin Xiong, and Xiaopeng Fan. 2024. Spiking tucker fusion transformer for audio-visual zero-shot learning. IEEE Transactions on Image Processing, 1--1.","journal-title":"IEEE Transactions on Image Processing, 1--1."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 16th European Conference on Computer Vision (ECCV), 121--137","author":"Xiujun","unstructured":"Xiujun Li et al. 2020. Oscar: object-semantics aligned pre-training for visionlanguage tasks. In Proceedings of the 16th European Conference on Computer Vision (ECCV), 121--137."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the 13th European Conference on Computer Vision (ECCV), 740--755","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: common objects in context. In Proceedings of the 13th European Conference on Computer Vision (ECCV), 740--755."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Fangyu Liu R\u00e9mi Lebret Didier Orel Philippe Sordet and Karl Aberer. 2020. Upgrading the newsroom: an automated image selection system for news articles. ACM Transactions on Multimedia Computing Communications and Applications (TOMM) 1--28.","DOI":"10.1145\/3396520"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00016"},{"key":"e_1_3_2_1_21_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Madry Aleksander","year":"2018","unstructured":"Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, and Adrian Vladu. 2018. Towards deep learning models resistant to adversarial attacks. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.17"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.282"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Konda Reddy Mopuri Utsav Garg and R Venkatesh Babu. 2017. Fast feature fool: a data independent approach to universal adversarial perturbations. In arXiv preprint arXiv:1707.05572.","DOI":"10.5244\/C.31.30"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_2"},{"key":"e_1_3_2_1_26_1","article-title":"Mra-net: improving vqa via multi-modal relation attention network","author":"Peng Liang","year":"2020","unstructured":"Liang Peng, Yang Yang, Zheng Wang, Zi Huang, and Heng Tao Shen. 2020. Mra-net: improving vqa via multi-modal relation attention network. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 318--329.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 318--329."},{"key":"e_1_3_2_1_27_1","unstructured":"Zhiliang Peng Li Dong Hangbo Bao Qixiang Ye and Furu Wei. 2022. BEiT v2: masked image modeling with vector-quantized visual tokenizers. In arXiv preprint arXiv:2208.06366."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00465"},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 8748--8763","author":"Alec","unstructured":"Alec Radford et al. 2021. Learning transferable visual models from natural language supervision. In Proceedings of the International Conference on Machine Learning (ICML), 8748--8763."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/1866696.1866717"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873987"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6017"},{"key":"e_1_3_2_1_34_1","article-title":"Visualizing data using t-sne","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-sne. Journal of Machine Learning Research (JMLR).","journal-title":"Journal of Machine Learning Research (JMLR)."},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML), 22680--22690","author":"Wang Teng","year":"2022","unstructured":"Teng Wang, Wenhao Jiang, Zhichao Lu, Feng Zheng, Ran Cheng, Chengguo Yin, and Ping Luo. 2022. Vlmixer: unpaired vision-language pre-training via cross-modal cutmix. In Proceedings of the International Conference on Machine Learning (ICML), 22680--22690."},{"key":"e_1_3_2_1_36_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Wenhui","unstructured":"Wenhui Wang et al. 2023. Image as a foreign language: BEiT pretraining for vision and vision-language tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), 19175--19186."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3249754"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01522"},{"key":"e_1_3_2_1_39_1","article-title":"Adaptive semi-supervised feature selection for cross-modal retrieval","author":"Yu En","year":"2018","unstructured":"En Yu, Jiande Sun, Jing Li, Xiaojun Chang, Xian-Hua Han, and Alexander G Hauptmann. 2018. Adaptive semi-supervised feature selection for cross-modal retrieval. IEEE Transactions on Multimedia (TMM), 1276--1288.","journal-title":"IEEE Transactions on Multimedia (TMM), 1276--1288."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00692"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547801"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Haonan Zheng Xinyang Deng Wen Jiang and Wenrui Li. 2024. A unified understanding of adversarial vulnerability regarding unimodal models and vision-language pre-training models. (2024). arXiv: 2407.17797 [cs.CV].","DOI":"10.1145\/3664647.3681184"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"e_1_3_2_1_45_1","volume-title":"Chen Change Loy, and Ziwei Liu","author":"Zhou Kaiyang","year":"2022","unstructured":"Kaiyang Zhou, Jingkang Yang, Chen Change Loy, and Ziwei Liu. 2022. Learning to prompt for vision-language models. International Journal of Computer Vision (IJCV), 2337--2348."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612454"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3686835","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3686835","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:28Z","timestamp":1750295848000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3686835"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":46,"alternative-id":["10.1145\/3664647.3686835","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3686835","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}