{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T16:58:05Z","timestamp":1778605085797,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":72,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62022009 and 61872021"],"award-info":[{"award-number":["62022009 and 61872021"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"The National Key Research and Development Plan of China","award":["2020AAA0103502"],"award-info":[{"award-number":["2020AAA0103502"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3547989","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:43:01Z","timestamp":1665416581000},"page":"5181-5190","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["Generating Transferable Adversarial Examples against Vision Transformers"],"prefix":"10.1145","author":[{"given":"Yuxuan","family":"Wang","sequence":"first","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiakai","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhongguancun Laboratory, State Key Lab of Software Development Environment, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zixin","family":"Yin","sequence":"additional","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruihao","family":"Gong","sequence":"additional","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, SenseTime, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingyi","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aishan","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xianglong","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Lab of Software Development Environment, Beihang University, Zhongguancun Laboratory, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Reveal of vision transformers robustness against adversarial attacks. arXiv preprint arXiv:2106.03734","author":"Aldahdooh Ahmed","year":"2021","unstructured":"Ahmed Aldahdooh , Wassim Hamidouche , and Olivier Deforges . 2021. Reveal of vision transformers robustness against adversarial attacks. arXiv preprint arXiv:2106.03734 ( 2021 ). Ahmed Aldahdooh, Wassim Hamidouche, and Olivier Deforges. 2021. Reveal of vision transformers robustness against adversarial attacks. arXiv preprint arXiv:2106.03734 (2021)."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_2_3_1","volume-title":"Are Transformers More Robust Than CNNs? arXiv: 2111.05464","author":"Bai Yutong","year":"2021","unstructured":"Yutong Bai , Jieru Mei , Alan L. Yuille , and Cihang Xie . 2021. Are Transformers More Robust Than CNNs? arXiv: 2111.05464 ( 2021 ). Yutong Bai, Jieru Mei, Alan L. Yuille, and Cihang Xie. 2021. Are Transformers More Robust Than CNNs? arXiv: 2111.05464 (2021)."},{"key":"e_1_3_2_2_4_1","volume-title":"Understanding Robustness of Transformers for Image Classification. arXiv:2103.14586","author":"Bhojanapalli Srinadh","year":"2021","unstructured":"Srinadh Bhojanapalli , Ayan Chakrabarti , Daniel Glasner , Daliang Li , Thomas Unterthiner , and Andreas Veit . 2021. Understanding Robustness of Transformers for Image Classification. arXiv:2103.14586 ( 2021 ). Srinadh Bhojanapalli, Ayan Chakrabarti, Daniel Glasner, Daliang Li, Thomas Unterthiner, and Andreas Veit. 2021. Understanding Robustness of Transformers for Image Classification. arXiv:2103.14586 (2021)."},{"key":"e_1_3_2_2_5_1","volume-title":"arXiv: 1712.09665","author":"Brown Tom B.","year":"2017","unstructured":"Tom B. Brown , Dandelion Man\u00e9 , Aurko Roy , Mart\u00edn Abadi , and Justin Gilmer . 2017. Adversarial Patch . arXiv: 1712.09665 ( 2017 ). Tom B. Brown, Dandelion Man\u00e9, Aurko Roy, Mart\u00edn Abadi, and Justin Gilmer. 2017. Adversarial Patch. arXiv: 1712.09665 (2017)."},{"key":"e_1_3_2_2_6_1","volume-title":"Transformer Interpretability Beyond Attention Visualization. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021","author":"Chefer Hila","year":"2021","unstructured":"Hila Chefer , Shir Gur , and Lior Wolf . 2021 . Transformer Interpretability Beyond Attention Visualization. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021 , virtual, June 19 --25 , 2021. Computer Vision Foundation \/ IEEE, 782--791. Hila Chefer, Shir Gur, and Lior Wolf. 2021. Transformer Interpretability Beyond Attention Visualization. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, virtual, June 19--25, 2021. Computer Vision Foundation \/ IEEE, 782--791."},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 357--366","author":"Richard Chen Chun-Fu","year":"2021","unstructured":"Chun-Fu Richard Chen , Quanfu Fan , and Rameswar Panda . 2021 . Crossvit: Crossattention multi-scale vision transformer for image classification . In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 357--366 . Chun-Fu Richard Chen, Quanfu Fan, and Rameswar Panda. 2021. Crossvit: Crossattention multi-scale vision transformer for image classification. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 357--366."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093610"},{"key":"e_1_3_2_2_9_1","volume-title":"Semantics Disentangling for Generalized Zero-Shot Learning. In IEEE\/CVF International Conference on Computer Vision (ICCV).","author":"Chen Zhi","year":"2021","unstructured":"Zhi Chen , Yadan Luo , Ruihong Qiu , Sen Wang , Zi Huang , Jingjing Li , and Zheng Zhang . 2021 . Semantics Disentangling for Generalized Zero-Shot Learning. In IEEE\/CVF International Conference on Computer Vision (ICCV). Zhi Chen, Yadan Luo, Ruihong Qiu, Sen Wang, Zi Huang, Jingjing Li, and Zheng Zhang. 2021. Semantics Disentangling for Generalized Zero-Shot Learning. In IEEE\/CVF International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the 28th ACM International Conference on Multimedia.","author":"Chen Zhi","year":"2021","unstructured":"Zhi Chen , Yadan Luo , Sen Wang , Ruihong Qiu , Jingjing Li , and Zi Huang . 2021 . Mitigating Generation Shifts for Generalized Zero-Shot Learning . In Proceedings of the 28th ACM International Conference on Multimedia. Zhi Chen, Yadan Luo, Sen Wang, Ruihong Qiu, Jingjing Li, and Zi Huang. 2021. Mitigating Generation Shifts for Generalized Zero-Shot Learning. In Proceedings of the 28th ACM International Conference on Multimedia."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413813"},{"key":"e_1_3_2_2_12_1","volume-title":"Attention-based models for speech recognition. Advances in neural information processing systems 28","author":"Chorowski Jan K","year":"2015","unstructured":"Jan K Chorowski , Dzmitry Bahdanau , Dmitriy Serdyuk , Kyunghyun Cho , and Yoshua Bengio . 2015. Attention-based models for speech recognition. Advances in neural information processing systems 28 ( 2015 ). Jan K Chorowski, Dzmitry Bahdanau, Dmitriy Serdyuk, Kyunghyun Cho, and Yoshua Bengio. 2015. Attention-based models for speech recognition. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18--24","volume":"2296","author":"Ascoli St\u00e9phane","year":"2021","unstructured":"St\u00e9phane d' Ascoli , Hugo Touvron , Matthew L. Leavitt , Ari S. Morcos , Giulio Biroli , and Levent Sagun . 2021 . ConViT: Improving Vision Transformers with Soft Convolutional Inductive Biases . In Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18--24 July 2021, Virtual Event (Proceedings of Machine Learning Research , Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 2286-- 2296 . http:\/\/proceedings.mlr.press\/v139\/d-ascoli21a.html St\u00e9phane d'Ascoli, Hugo Touvron, Matthew L. Leavitt, Ari S. Morcos, Giulio Biroli, and Levent Sagun. 2021. ConViT: Improving Vision Transformers with Soft Convolutional Inductive Biases. In Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18--24 July 2021, Virtual Event (Proceedings of Machine Learning Research, Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 2286--2296. http:\/\/proceedings.mlr.press\/v139\/d-ascoli21a.html"},{"key":"e_1_3_2_2_14_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018). Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_15_1","volume-title":"Where and How to Transfer: Knowledge Aggregation-Induced Transferability Perception for Unsupervised Domain Adaptation","author":"Dong Jiahua","year":"2021","unstructured":"Jiahua Dong , Yang Cong , Gan Sun , Zhen Fang , and Zhengming Ding . 2021. Where and How to Transfer: Knowledge Aggregation-Induced Transferability Perception for Unsupervised Domain Adaptation . IEEE Transactions on Pattern Analysis and Machine Intelligence ( 2021 ), 1--1. https:\/\/doi.org\/10.1109\/TPAMI. 2021.3128560 Jiahua Dong, Yang Cong, Gan Sun, Zhen Fang, and Zhengming Ding. 2021. Where and How to Transfer: Knowledge Aggregation-Induced Transferability Perception for Unsupervised Domain Adaptation. IEEE Transactions on Pattern Analysis and Machine Intelligence (2021), 1--1. https:\/\/doi.org\/10.1109\/TPAMI. 2021.3128560"},{"key":"e_1_3_2_2_16_1","volume-title":"What Can Be Transferred: Unsupervised Domain Adaptation for Endoscopic Lesions Segmentation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4022--4031","author":"Dong Jiahua","year":"2020","unstructured":"Jiahua Dong , Yang Cong , Gan Sun , Bineng Zhong , and Xiaowei Xu . 2020 . What Can Be Transferred: Unsupervised Domain Adaptation for Endoscopic Lesions Segmentation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4022--4031 . Jiahua Dong, Yang Cong, Gan Sun, Bineng Zhong, and Xiaowei Xu. 2020. What Can Be Transferred: Unsupervised Domain Adaptation for Endoscopic Lesions Segmentation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4022--4031."},{"key":"e_1_3_2_2_17_1","volume-title":"ICLR","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy , Lucas Beyer , Alexander Kolesnikov , Dirk Weissenborn , Xiaohua Zhai , Thomas Unterthiner , Mostafa Dehghani , Matthias Minderer , Georg Heigold , Sylvain Gelly , Jakob Uszkoreit , and Neil Houlsby . 2021 . An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale . In ICLR 2021. Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In ICLR 2021."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01580"},{"key":"e_1_3_2_2_19_1","volume-title":"International Conference on Learning Representations. https: \/\/openreview.net\/forum?id=28ib9tf6zhr","author":"Fu Yonggan","year":"2022","unstructured":"Yonggan Fu , Shunyao Zhang , Shang Wu , Cheng Wan , and Yingyan Lin . 2022 . Patch-Fool: Are Vision Transformers Always Robust Against Adversarial Perturbations? . In International Conference on Learning Representations. https: \/\/openreview.net\/forum?id=28ib9tf6zhr Yonggan Fu, Shunyao Zhang, Shang Wu, Cheng Wan, and Yingyan Lin. 2022. Patch-Fool: Are Vision Transformers Always Robust Against Adversarial Perturbations?. In International Conference on Learning Representations. https: \/\/openreview.net\/forum?id=28ib9tf6zhr"},{"key":"e_1_3_2_2_20_1","volume-title":"Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572","author":"Goodfellow Ian J","year":"2014","unstructured":"Ian J Goodfellow , Jonathon Shlens , and Christian Szegedy . 2014. Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572 ( 2014 ). Ian J Goodfellow, Jonathon Shlens, and Christian Szegedy. 2014. Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572 (2014)."},{"key":"e_1_3_2_2_21_1","volume-title":"Are Vision Transformers Robust to Patch Perturbations? arXiv: 2111.10659","author":"Gu Jindong","year":"2021","unstructured":"Jindong Gu , Volker Tresp , and Yao Qin . 2021. Are Vision Transformers Robust to Patch Perturbations? arXiv: 2111.10659 ( 2021 ). Jindong Gu, Volker Tresp, and Yao Qin. 2021. Are Vision Transformers Robust to Patch Perturbations? arXiv: 2111.10659 (2021)."},{"key":"e_1_3_2_2_22_1","volume-title":"Attention Mechanisms in Computer Vision: A Survey. arXiv: 2111.07624","author":"Guo Meng-Hao","year":"2021","unstructured":"Meng-Hao Guo , Tian-Xing Xu , Jiangjiang Liu , Zheng-Ning Liu , Peng-Tao Jiang , Tai-Jiang Mu , Song-Hai Zhang , Ralph R. Martin , Ming-Ming Cheng , and Shi-Min Hu. 2021. Attention Mechanisms in Computer Vision: A Survey. arXiv: 2111.07624 ( 2021 ). Meng-Hao Guo, Tian-Xing Xu, Jiangjiang Liu, Zheng-Ning Liu, Peng-Tao Jiang, Tai-Jiang Mu, Song-Hai Zhang, Ralph R. Martin, Ming-Ming Cheng, and Shi-Min Hu. 2021. Attention Mechanisms in Computer Vision: A Survey. arXiv: 2111.07624 (2021)."},{"key":"e_1_3_2_2_23_1","volume-title":"A Survey on Visual Transformer. arXiv","author":"Han Kai","year":"2012","unstructured":"Kai Han , Yunhe Wang , Hanting Chen , Xinghao Chen , Jianyuan Guo , Zhenhua Liu , Yehui Tang , An Xiao , Chunjing Xu , Yixing Xu , Zhaohui Yang , Yiman Zhang , and Dacheng Tao . 2020. A Survey on Visual Transformer. arXiv : 2012 .12556 (2020). Kai Han, Yunhe Wang, Hanting Chen, Xinghao Chen, Jianyuan Guo, Zhenhua Liu, Yehui Tang, An Xiao, Chunjing Xu, Yixing Xu, Zhaohui Yang, Yiman Zhang, and Dacheng Tao. 2020. A Survey on Visual Transformer. arXiv: 2012.12556 (2020)."},{"key":"e_1_3_2_2_24_1","volume-title":"Transformer in transformer. Advances in Neural Information Processing Systems 34","author":"Han Kai","year":"2021","unstructured":"Kai Han , An Xiao , Enhua Wu , Jianyuan Guo , Chunjing Xu , and Yunhe Wang . 2021. Transformer in transformer. Advances in Neural Information Processing Systems 34 ( 2021 ). Kai Han, An Xiao, Enhua Wu, Jianyuan Guo, Chunjing Xu, and Yunhe Wang. 2021. Transformer in transformer. Advances in Neural Information Processing Systems 34 (2021)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"e_1_3_2_2_27_1","volume-title":"Adversarial Token Attacks on Vision Transformers. arXiv:2110.04337","author":"Joshi Ameya","year":"2021","unstructured":"Ameya Joshi , Gauri Jagatap , and Chinmay Hegde . 2021. Adversarial Token Attacks on Vision Transformers. arXiv:2110.04337 ( 2021 ). Ameya Joshi, Gauri Jagatap, and Chinmay Hegde. 2021. Adversarial Token Attacks on Vision Transformers. arXiv:2110.04337 (2021)."},{"key":"e_1_3_2_2_28_1","volume-title":"One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997","author":"Krizhevsky Alex","year":"2014","unstructured":"Alex Krizhevsky . 2014. One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997 ( 2014 ). Alex Krizhevsky. 2014. One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997 (2014)."},{"key":"e_1_3_2_2_29_1","volume-title":"Adversarial examples in the physical world. arXiv preprint arXiv:1607.02533","author":"Kurakin Alexey","year":"2016","unstructured":"Alexey Kurakin , Ian Goodfellow , and Samy Bengio . 2016. Adversarial examples in the physical world. arXiv preprint arXiv:1607.02533 ( 2016 ). Alexey Kurakin, Ian Goodfellow, and Samy Bengio. 2016. Adversarial examples in the physical world. arXiv preprint arXiv:1607.02533 (2016)."},{"key":"e_1_3_2_2_30_1","unstructured":"Maosen Li Yanhua Yang Kun Wei Xu Yang and Heng Huang. 2022. Learning Universal Adversarial Perturbation by Adversarial Example. (2022).  Maosen Li Yanhua Yang Kun Wei Xu Yang and Heng Huang. 2022. Learning Universal Adversarial Perturbation by Adversarial Example. (2022)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58574-7_3"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Siyuan Liang Baoyuan Wu Yanbo Fan Xingxing Wei and Xiaochun Cao. 2021. Parallel rectangle flip attack: A query-based black-box attack against object detection. In ICCV.  Siyuan Liang Baoyuan Wu Yanbo Fan Xingxing Wei and Xiaochun Cao. 2021. Parallel rectangle flip attack: A query-based black-box attack against object detection. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00760"},{"key":"e_1_3_2_2_33_1","volume-title":"Parallel rectangle flip attack: A query-based black-box attack against object detection. arXiv preprint arXiv:2201.08970","author":"Liang Siyuan","year":"2022","unstructured":"Siyuan Liang , Baoyuan Wu , Yanbo Fan , Xingxing Wei , and Xiaochun Cao . 2022. Parallel rectangle flip attack: A query-based black-box attack against object detection. arXiv preprint arXiv:2201.08970 ( 2022 ). Siyuan Liang, Baoyuan Wu, Yanbo Fan, Xingxing Wei, and Xiaochun Cao. 2022. Parallel rectangle flip attack: A query-based black-box attack against object detection. arXiv preprint arXiv:2201.08970 (2022)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58520-4_8"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33011028"},{"key":"e_1_3_2_2_36_1","unstructured":"Aishan Liu Jiakai Wang Xianglong Liu Bowen Cao Chongzhi Zhang and Hang Yu. 2020. Bias-based universal adversarial patch attack for automatic check-out. In ECCV.  Aishan Liu Jiakai Wang Xianglong Liu Bowen Cao Chongzhi Zhang and Hang Yu. 2020. Bias-based universal adversarial patch attack for automatic check-out. In ECCV."},{"key":"e_1_3_2_2_37_1","volume-title":"A Survey of Visual Transformers. arXiv: 2111.06091","author":"Liu Yang","year":"2021","unstructured":"Yang Liu , Yao Zhang , Yixin Wang , Feng Hou , Jin Yuan , Jiang Tian , Yang Zhang , Zhongchao Shi , Jianping Fan , and Zhiqiang He. 2021. A Survey of Visual Transformers. arXiv: 2111.06091 ( 2021 ). Yang Liu, Yao Zhang, Yixin Wang, Feng Hou, Jin Yuan, Jiang Tian, Yang Zhang, Zhongchao Shi, Jianping Fan, and Zhiqiang He. 2021. A Survey of Visual Transformers. arXiv: 2111.06091 (2021)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_39_1","volume-title":"Towards deep learning models resistant to adversarial attacks. arXiv preprint arXiv:1706.06083","author":"Madry Aleksander","year":"2017","unstructured":"Aleksander Madry , Aleksandar Makelov , Ludwig Schmidt , Dimitris Tsipras , and Adrian Vladu . 2017. Towards deep learning models resistant to adversarial attacks. arXiv preprint arXiv:1706.06083 ( 2017 ). Aleksander Madry, Aleksandar Makelov, Ludwig Schmidt, Dimitris Tsipras, and Adrian Vladu. 2017. Towards deep learning models resistant to adversarial attacks. arXiv preprint arXiv:1706.06083 (2017)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00774"},{"key":"e_1_3_2_2_41_1","volume-title":"On the Robustness of Vision Transformers to Adversarial Examples. arXiv: 2104.02610","author":"Mahmood Kaleel","year":"2021","unstructured":"Kaleel Mahmood , Rigel Mahmood , and Marten van Dijk . 2021. On the Robustness of Vision Transformers to Adversarial Examples. arXiv: 2104.02610 ( 2021 ). Kaleel Mahmood, Rigel Mahmood, and Marten van Dijk. 2021. On the Robustness of Vision Transformers to Adversarial Examples. arXiv: 2104.02610 (2021)."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Seyed-Mohsen Moosavi-Dezfooli Alhussein Fawzi and Pascal Frossard. 2016. DeepFool: A Simple and Accurate Method to Fool Deep Neural Networks. In CVPR.  Seyed-Mohsen Moosavi-Dezfooli Alhussein Fawzi and Pascal Frossard. 2016. DeepFool: A Simple and Accurate Method to Fool Deep Neural Networks. In CVPR.","DOI":"10.1109\/CVPR.2016.282"},{"key":"e_1_3_2_2_43_1","volume-title":"Fahad Shahbaz Khan, and Fatih Porikli","author":"Naseer Muzammal","year":"2021","unstructured":"Muzammal Naseer , Kanchana Ranasinghe , Salman Khan , Fahad Shahbaz Khan, and Fatih Porikli . 2021 . On improving adversarial transferability of vision transformers. arXiv preprint arXiv:2106.04169 (2021). Muzammal Naseer, Kanchana Ranasinghe, Salman Khan, Fahad Shahbaz Khan, and Fatih Porikli. 2021. On improving adversarial transferability of vision transformers. arXiv preprint arXiv:2106.04169 (2021)."},{"key":"e_1_3_2_2_44_1","volume-title":"Fahad Shahbaz Khan, and Ming-Hsuan Yang","author":"Naseer Muzammal","year":"2021","unstructured":"Muzammal Naseer , Kanchana Ranasinghe , Salman H. Khan , Munawar Hayat , Fahad Shahbaz Khan, and Ming-Hsuan Yang . 2021 . Intriguing Properties of Vision Transformers . arXiv: 2105.10497 (2021). Muzammal Naseer, Kanchana Ranasinghe, Salman H. Khan, Munawar Hayat, Fahad Shahbaz Khan, and Ming-Hsuan Yang. 2021. Intriguing Properties of Vision Transformers. arXiv: 2105.10497 (2021)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3052973.3053009"},{"key":"e_1_3_2_2_46_1","volume-title":"Vision Transformers are Robust Learners. arXiv: 2105.07581","author":"Paul Sayak","year":"2021","unstructured":"Sayak Paul and Pin-Yu Chen . 2021. Vision Transformers are Robust Learners. arXiv: 2105.07581 ( 2021 ). Sayak Paul and Pin-Yu Chen. 2021. Vision Transformers are Robust Learners. arXiv: 2105.07581 (2021)."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_2_48_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 ( 2014 ). Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2019.2890858"},{"key":"e_1_3_2_2_51_1","volume-title":"Intriguing properties of neural networks. arXiv preprint arXiv:1312.6199","author":"Szegedy Christian","year":"2013","unstructured":"Christian Szegedy , Wojciech Zaremba , Ilya Sutskever , Joan Bruna , Dumitru Erhan , Ian Goodfellow , and Rob Fergus . 2013. Intriguing properties of neural networks. arXiv preprint arXiv:1312.6199 ( 2013 ). Christian Szegedy, Wojciech Zaremba, Ilya Sutskever, Joan Bruna, Dumitru Erhan, Ian Goodfellow, and Rob Fergus. 2013. Intriguing properties of neural networks. arXiv preprint arXiv:1312.6199 (2013)."},{"key":"e_1_3_2_2_52_1","volume-title":"Robustart: Benchmarking robustness on architecture design and training techniques. arXiv preprint arXiv:2109.05211","author":"Tang Shiyu","year":"2021","unstructured":"Shiyu Tang , Ruihao Gong , Yan Wang , Aishan Liu , Jiakai Wang , Xinyun Chen , Fengwei Yu , Xianglong Liu , Dawn Song , Alan Yuille , 2021 . Robustart: Benchmarking robustness on architecture design and training techniques. arXiv preprint arXiv:2109.05211 (2021). Shiyu Tang, Ruihao Gong, Yan Wang, Aishan Liu, Jiakai Wang, Xinyun Chen, Fengwei Yu, Xianglong Liu, Dawn Song, Alan Yuille, et al. 2021. Robustart: Benchmarking robustness on architecture design and training techniques. arXiv preprint arXiv:2109.05211 (2021)."},{"key":"e_1_3_2_2_53_1","volume-title":"Wiebe Van Ranst, and Toon Goedem\u00e9","author":"Thys Simen","year":"2019","unstructured":"Simen Thys , Wiebe Van Ranst, and Toon Goedem\u00e9 . 2019 . Fooling automated surveillance cameras: adversarial patches to attack person detection. In CVPRW. Simen Thys, Wiebe Van Ranst, and Toon Goedem\u00e9. 2019. Fooling automated surveillance cameras: adversarial patches to attack person detection. In CVPRW."},{"key":"e_1_3_2_2_54_1","volume-title":"Training data-efficient image transformers & distillation through attention. arXiv","author":"Touvron Hugo","year":"2012","unstructured":"Hugo Touvron , Matthieu Cord , Matthijs Douze , Francisco Massa , Alexandre Sablayrolles , and Herv\u00e9 J\u00e9gou . 2020. Training data-efficient image transformers & distillation through attention. arXiv : 2012 .12877 (2020). Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herv\u00e9 J\u00e9gou. 2020. Training data-efficient image transformers & distillation through attention. arXiv: 2012.12877 (2020)."},{"key":"e_1_3_2_2_55_1","article-title":"Visualizing data using t-SNE","volume":"9","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton . 2008 . Visualizing data using t-SNE . Journal of machine learning research 9 , 11 (2008). Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. Journal of machine learning research 9, 11 (2008).","journal-title":"Journal of machine learning research"},{"key":"e_1_3_2_2_56_1","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N. Gomez , Lukasz Kaiser , and Illia Polosukhin . 2017 . Attention is All you Need . In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017. 5998--6008. Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017. 5998--6008."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3127849"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00846"},{"key":"e_1_3_2_2_59_1","volume-title":"NonLocal Neural Networks. In CVPR","author":"Wang Xiaolong","year":"2018","unstructured":"Xiaolong Wang , Ross B. Girshick , Abhinav Gupta , and Kaiming He . 2018 . NonLocal Neural Networks. In CVPR 2018. 7794--7803. Xiaolong Wang, Ross B. Girshick, Abhinav Gupta, and Kaiming He. 2018. NonLocal Neural Networks. In CVPR 2018. 7794--7803."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"crossref","unstructured":"Kun Wei Cheng Deng Xu Yang etal 2020. Lifelong Zero-Shot Learning.. In IJCAI. 551--557.  Kun Wei Cheng Deng Xu Yang et al. 2020. Lifelong Zero-Shot Learning.. In IJCAI. 551--557.","DOI":"10.24963\/ijcai.2020\/77"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3110369"},{"key":"e_1_3_2_2_62_1","volume-title":"Transferable adversarial attacks for image and video object detection. arXiv preprint arXiv:1811.12641","author":"Wei Xingxing","year":"2018","unstructured":"Xingxing Wei , Siyuan Liang , Ning Chen , and Xiaochun Cao . 2018. Transferable adversarial attacks for image and video object detection. arXiv preprint arXiv:1811.12641 ( 2018 ). Xingxing Wei, Siyuan Liang, Ning Chen, and Xiaochun Cao. 2018. Transferable adversarial attacks for image and video object detection. arXiv preprint arXiv:1811.12641 (2018)."},{"key":"e_1_3_2_2_63_1","volume-title":"Towards Transferable Adversarial Attacks on Vision Transformers. arXiv: 2109.04176","author":"Wei Zhipeng","year":"2021","unstructured":"Zhipeng Wei , Jingjing Chen , Micah Goldblum , Zuxuan Wu , Tom Goldstein , and Yu-Gang Jiang . 2021. Towards Transferable Adversarial Attacks on Vision Transformers. arXiv: 2109.04176 ( 2021 ). Zhipeng Wei, Jingjing Chen, Micah Goldblum, Zuxuan Wu, Tom Goldstein, and Yu-Gang Jiang. 2021. Towards Transferable Adversarial Attacks on Vision Transformers. arXiv: 2109.04176 (2021)."},{"key":"e_1_3_2_2_64_1","unstructured":"Ross Wightman. 2019. PyTorch Image Models. https:\/\/github.com\/rwightman\/pytorch-image-models. https:\/\/doi.org\/10.5281\/zenodo.4414861  Ross Wightman. 2019. PyTorch Image Models. https:\/\/github.com\/rwightman\/pytorch-image-models. https:\/\/doi.org\/10.5281\/zenodo.4414861"},{"key":"e_1_3_2_2_65_1","first-page":"9098","article-title":"Adversarial learning for robust deep clustering","volume":"33","author":"Yang Xu","year":"2020","unstructured":"Xu Yang , Cheng Deng , Kun Wei , Junchi Yan , and Wei Liu . 2020 . Adversarial learning for robust deep clustering . Advances in Neural Information Processing Systems 33 (2020), 9098 -- 9108 . Xu Yang, Cheng Deng, Kun Wei, Junchi Yan, and Wei Liu. 2020. Adversarial learning for robust deep clustering. Advances in Neural Information Processing Systems 33 (2020), 9098--9108.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/135"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2020.107734"},{"key":"e_1_3_2_2_69_1","volume-title":"Efficient and Model-Based Infrared and Visible Image Fusion via Algorithm Unrolling","author":"Zhao Zixiang","year":"2021","unstructured":"Zixiang Zhao , Shuang Xu , Jiangshe Zhang , Chengyang Liang , Chunxia Zhang , and Junmin Liu . 2021. Efficient and Model-Based Infrared and Visible Image Fusion via Algorithm Unrolling . IEEE Transactions on Circuits and Systems for Video Technology ( 2021 ). Zixiang Zhao, Shuang Xu, Jiangshe Zhang, Chengyang Liang, Chunxia Zhang, and Junmin Liu. 2021. Efficient and Model-Based Infrared and Visible Image Fusion via Algorithm Unrolling. IEEE Transactions on Circuits and Systems for Video Technology (2021)."},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME51207.2021.9428272"},{"key":"e_1_3_2_2_71_1","volume-title":"Invisible mask: Practical attacks on face recognition with infrared. arXiv preprint arXiv:1803.04683","author":"Zhou Zhe","year":"2018","unstructured":"Zhe Zhou , Di Tang , Xiaofeng Wang , Weili Han , Xiangyu Liu , and Kehuan Zhang . 2018. Invisible mask: Practical attacks on face recognition with infrared. arXiv preprint arXiv:1803.04683 ( 2018 ). Zhe Zhou, Di Tang, Xiaofeng Wang, Weili Han, Xiangyu Liu, and Kehuan Zhang. 2018. Invisible mask: Practical attacks on face recognition with infrared. arXiv preprint arXiv:1803.04683 (2018)."},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16477"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547989","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3547989","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:00:31Z","timestamp":1750186831000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547989"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":72,"alternative-id":["10.1145\/3503161.3547989","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3547989","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}