{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T18:24:01Z","timestamp":1772907841900,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","funder":[{"name":"Hong Kong Innovation and Technology Commission","award":["InnoHK Project CIMDA"],"award-info":[{"award-number":["InnoHK Project CIMDA"]}]},{"DOI":"10.13039\/100007567","name":"City University of Hong Kong","doi-asserted-by":"publisher","award":["9610034, 9610460"],"award-info":[{"award-number":["9610034, 9610460"]}],"id":[{"id":"10.13039\/100007567","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Natural Science Special Project Research Fund of Guizhou University","award":["2025-06"],"award-info":[{"award-number":["2025-06"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754825","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:56:44Z","timestamp":1761375404000},"page":"9140-9149","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["ELFATT: Efficient Linear Fast Attention for Vision Transformers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3405-742X","authenticated-orcid":false,"given":"Chong","family":"Wu","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering and Centre for Intelligent Multidimensional Data Analysis, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9956-062X","authenticated-orcid":false,"given":"Maolin","family":"Che","sequence":"additional","affiliation":[{"name":"School of Mathematics and Statistics and State Key Laboratory of Public Big Data, Guizhou University, Guiyang, China and Department of Electrical Engineering and Centre for Intelligent Multidimensional Data Analysis, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6951-7273","authenticated-orcid":false,"given":"Renjie","family":"Xu","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Centre for Intelligent Multidimensional Data Analysis, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0446-6276","authenticated-orcid":false,"given":"Zhuoheng","family":"Ran","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Centre for Intelligent Multidimensional Data Analysis, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9661-3095","authenticated-orcid":false,"given":"Hong","family":"Yan","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Centre for Intelligent Multidimensional Data Analysis, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Longformer: The long-document transformer. CoRR","author":"Beltagy Iz","year":"2020","unstructured":"Iz Beltagy, Matthew E. Peters, and Arman Cohan. 2020. Longformer: The long-document transformer. CoRR, Vol. abs\/2004.05150 (2020). arXiv:2004.05150 https:\/\/arxiv.org\/abs\/2004.05150"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-25082-8_3"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00484"},{"key":"e_1_3_2_1_5_1","volume-title":"Video generation models as world simulators. OpenAI","author":"Brooks Tim","year":"2024","unstructured":"Tim Brooks, Bill Peebles, Connor Holmes, Will DePue, Yufei Guo, Li Jing, David Schnurr, Joe Taylor, Troy Luhman, Eric Luhman, Clarence Ng, Ricky Wang, and Aditya Ramesh. 2024. Video generation models as world simulators. OpenAI (2024). https:\/\/openai.com\/index\/video-generation-models-as-world-simulators"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01587"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01157"},{"key":"e_1_3_2_1_8_1","first-page":"65088","article-title":"Primal-Attention: Self-attention through asymmetric kernel SVD in primal representation. In Advances in Neural Information Processing Systems, A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Eds.), Vol. 36. Curran Associates","author":"Chen Yingyi","year":"2023","unstructured":"Yingyi Chen, Qinghua Tao, Francesco Tonin, and Johan Suykens. 2023b. Primal-Attention: Self-attention through asymmetric kernel SVD in primal representation. In Advances in Neural Information Processing Systems, A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Eds.), Vol. 36. Curran Associates, Inc., 65088-65101.","journal-title":"Inc."},{"key":"e_1_3_2_1_9_1","volume-title":"Generating long sequences with sparse transformers. CoRR","author":"Child Rewon","year":"2019","unstructured":"Rewon Child, Scott Gray, Alec Radford, and Ilya Sutskever. 2019. Generating long sequences with sparse transformers. CoRR, Vol. abs\/1904.10509 (2019). arXiv:1904.10509 http:\/\/arxiv.org\/abs\/1904.10509"},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Ua6zuk0WRH","author":"Choromanski Krzysztof Marcin","year":"2021","unstructured":"Krzysztof Marcin Choromanski, Valerii Likhosherstov, David Dohan, Xingyou Song, Andreea Gane, Tamas Sarlos, Peter Hawkins, Jared Quincy Davis, Afroz Mohiuddin, Lukasz Kaiser, David Benjamin Belanger, Lucy J Colwell, and Adrian Weller. 2021. Rethinking attention with performers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Ua6zuk0WRH"},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=3KWnuT-R1bh","author":"Chu Xiangxiang","year":"2023","unstructured":"Xiangxiang Chu, Zhi Tian, Bo Zhang, Xinlong Wang, and Chunhua Shen. 2023. Conditional positional encodings for vision transformers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=3KWnuT-R1bh"},{"key":"e_1_3_2_1_12_1","volume-title":"Fast and accurate deep network learning by exponential linear units (ELUs). arXiv preprint arXiv:1511.07289","author":"Clevert Djork-Arn\u00e9","year":"2015","unstructured":"Djork-Arn\u00e9 Clevert. 2015. Fast and accurate deep network learning by exponential linear units (ELUs). arXiv preprint arXiv:1511.07289 (2015)."},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=mZn2Xyh9Ec","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster attention with better parallelism and work partitioning. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=mZn2Xyh9Ec"},{"key":"e_1_3_2_1_14_1","first-page":"16344","volume-title":"Oh (Eds.)","volume":"35","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Dan Fu, Stefano Ermon, Atri Rudra, and Christopher R'e. 2022. FlashAttention: Fast and memory-efficient exact attention with IO-awareness. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 16344-16359."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"10071","author":"Dao Tri","year":"2024","unstructured":"Tri Dao and Albert Gu. 2024. Transformers are SSMs: Generalized models and efficient algorithms through structured state space duality. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235), Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp (Eds.). PMLR, 10041-10071."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"e_1_3_2_1_17_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"First Conference on Language Modeling. https:\/\/openreview.net\/forum?id=tEYskw1VY2","author":"Gu Albert","year":"2024","unstructured":"Albert Gu and Tri Dao. 2024. Mamba: Linear-time sequence modeling with selective state spaces. In First Conference on Language Modeling. https:\/\/openreview.net\/forum?id=tEYskw1VY2"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00548"},{"key":"e_1_3_2_1_20_1","volume-title":"Agent Attention: On the integration of softmax and linear attention. In Computer Vision - ECCV","author":"Han Dongchen","year":"2025","unstructured":"Dongchen Han, Tianzhu Ye, Yizeng Han, Zhuofan Xia, Siyuan Pan, Pengfei Wan, Shiji Song, and Gao Huang. 2025. Agent Attention: On the integration of softmax and linear attention. In Computer Vision - ECCV 2024, Ale\u0161 Leonardis, Elisa Ricci, Stefan Roth, Olga Russakovsky, Torsten Sattler, and G\u00fcl Varol (Eds.). Springer Nature Switzerland, Cham, 124-140."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19962"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_1_23_1","volume-title":"Advances in Neural Information Processing Systems","author":"Heusel Martin","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. GANs trained by a two time-scale update rule converge to a local Nash equilibrium. In Advances in Neural Information Processing Systems, I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett (Eds.), Vol. 30. Curran Associates, Inc."},{"key":"e_1_3_2_1_24_1","volume-title":"Interactive multi-head self-attention with linear complexity. arXiv preprint arXiv:2402.17507","author":"Kang Hankyul","year":"2024","unstructured":"Hankyul Kang, Ming-Hsuan Yang, and Jongbin Ryu. 2024. Interactive multi-head self-attention with linear complexity. arXiv preprint arXiv:2402.17507 (2024)."},{"key":"e_1_3_2_1_25_1","first-page":"5156","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research","author":"Katharopoulos Angelos","year":"2020","unstructured":"Angelos Katharopoulos, Apoorv Vyas, Nikolaos Pappas, and Fran\u00e7ois Fleuret. 2020. Transformers are RNNs: Fast autoregressive transformers with linear attention. In Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 119), Hal Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 5156-5165."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_27_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkgNKkHtvB","author":"Kitaev Nikita","year":"2020","unstructured":"Nikita Kitaev, Lukasz Kaiser, and Anselm Levskaya. 2020. Reformer: The efficient transformer. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkgNKkHtvB"},{"key":"e_1_3_2_1_28_1","volume-title":"Microsoft COCO: Common objects in context. arXiv preprint arXiv:1405.0312","author":"Lin Tsung-Yi","year":"2015","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, Lubomir Bourdev, Ross Girshick, James Hays, Pietro Perona, Deva Ramanan, C. Lawrence Zitnick, and Piotr Doll\u00e1r. 2015. Microsoft COCO: Common objects in context. arXiv preprint arXiv:1405.0312 (2015)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01386"},{"key":"e_1_3_2_1_30_1","first-page":"103031","volume-title":"Zhang (Eds.)","volume":"37","author":"Liu Yue","year":"2024","unstructured":"Yue Liu, Yunjie Tian, Yuzhong Zhao, Hongtian Yu, Lingxi Xie, Yaowei Wang, Qixiang Ye, Jianbin Jiao, and Yunfan Liu. 2024. VMamba: Visual state space model. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang (Eds.), Vol. 37. Curran Associates, Inc., 103031-103063."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"e_1_3_2_1_33_1","first-page":"21297","volume-title":"Wortman Vaughan (Eds.)","volume":"34","author":"Lu Jiachen","year":"2021","unstructured":"Jiachen Lu, Jinghan Yao, Junge Zhang, Xiatian Zhu, Hang Xu, Weiguo Gao, Chunjing Xu, Tao Xiang, and Li Zhang. 2021. SOFT: Softmax-free transformer with linear complexity. In Advances in Neural Information Processing Systems, M. Ranzato, A. Beygelzimer, Y. Dauphin, P.S. Liang, and J. Wortman Vaughan (Eds.), Vol. 34. Curran Associates, Inc., 21297-21309."},{"key":"e_1_3_2_1_34_1","first-page":"807","volume-title":"Proceedings of the 27th International Conference on Machine Learning","author":"Nair Vinod","unstructured":"Vinod Nair and Geoffrey E. Hinton. 2010. Rectified linear units improve restricted boltzmann machines. In Proceedings of the 27th International Conference on Machine Learning (Haifa, Israel) (ICML'10). Omnipress, Madison, WI, USA, 807-814."},{"key":"e_1_3_2_1_35_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=QtTKTdVrFBB","author":"Peng Hao","year":"2021","unstructured":"Hao Peng, Nikolaos Pappas, Dani Yogatama, Roy Schwartz, Noah Smith, and Lingpeng Kong. 2021. Random feature attention. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=QtTKTdVrFBB"},{"key":"e_1_3_2_1_36_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bl8CQrx2Up4","author":"Qin Zhen","year":"2022","unstructured":"Zhen Qin, Weixuan Sun, Hui Deng, Dongxu Li, Yunshen Wei, Baohong Lv, Junjie Yan, Lingpeng Kong, and Yiran Zhong. 2022. cosFormer: Rethinking softmax in attention. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bl8CQrx2Up4"},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Zhdhg6n2OG","author":"Ramapuram Jason","year":"2025","unstructured":"Jason Ramapuram, Federico Danieli, Eeshan Gunesh Dhekane, Floris Weers, Dan Busbridge, Pierre Ablin, Tatiana Likhomanenko, Jagrit Digani, Zijin Gu, Amitis Shidani, and Russell Webb. 2025. Theory, analysis, and best practices for sigmoid self-attention. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Zhdhg6n2OG"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS56072.2025.11043624"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_41_1","volume-title":"Advances in Neural Information Processing Systems","author":"Shah Jay","unstructured":"Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, and Tri Dao. 2024. FlashAttention-3: Fast and accurate attention with asynchrony and low-precision. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang (Eds.), Vol. 37. Curran Associates, Inc., 68658-68685."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2074"},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV). 3531-3539","author":"Shen Zhuoran","year":"2021","unstructured":"Zhuoran Shen, Mingyuan Zhang, Haiyu Zhao, Shuai Yi, and Hongsheng Li. 2021. Efficient Attention: Attention with linear complexities. In Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV). 3531-3539."},{"key":"e_1_3_2_1_44_1","first-page":"9438","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research","author":"Tay Yi","year":"2020","unstructured":"Yi Tay, Dara Bahri, Liu Yang, Donald Metzler, and Da-Cheng Juan. 2020. Sparse sinkhorn attention. In Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 119), Hal Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 9438-9447."},{"key":"e_1_3_2_1_45_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=qVyeW-grC2k","author":"Tay Yi","year":"2021","unstructured":"Yi Tay, Mostafa Dehghani, Samira Abnar, Yikang Shen, Dara Bahri, Philip Pham, Jinfeng Rao, Liu Yang, Sebastian Ruder, and Donald Metzler. 2021. Long Range Arena : A benchmark for efficient transformers. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=qVyeW-grC2k"},{"key":"e_1_3_2_1_46_1","volume-title":"\u0141 ukasz Kaiser, and Illia Polosukhin","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems, I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett (Eds.), Vol. 30. Curran Associates, Inc."},{"key":"e_1_3_2_1_47_1","volume-title":"Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768","author":"Wang Sinong","year":"2020","unstructured":"Sinong Wang, Belinda Z Li, Madian Khabsa, Han Fang, and Hao Ma. 2020. Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)."},{"key":"e_1_3_2_1_48_1","volume-title":"Replacing softmax with ReLU in vision transformers. arXiv preprint arXiv:2309.08586","author":"Wortsman Mitchell","year":"2023","unstructured":"Mitchell Wortsman, Jaehoon Lee, Justin Gilmer, and Simon Kornblith. 2023. Replacing softmax with ReLU in vision transformers. arXiv preprint arXiv:2309.08586 (2023)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.36227\/techrxiv.171392846.60982484\/v3"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00988"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1080\/01630563.2024.2318602"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40314-023-02565-7"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"crossref","first-page":"185","DOI":"10.61208\/pjo-2023-042","article-title":"Randomized quaternion matrix UTV decomposition and its applications in quaternion matrix optimization","volume":"20","author":"Xu Renjie","year":"2024","unstructured":"Renjie Xu and Yimin Wei. 2024. Randomized quaternion matrix UTV decomposition and its applications in quaternion matrix optimization. Pacific Journal of Optimization, Vol. 20, 2 (2024), 185-211.","journal-title":"Pacific Journal of Optimization"},{"key":"e_1_3_2_1_56_1","volume-title":"Multi-scale context aggregation by dilated convolutions. arXiv preprint arXiv:1511.07122","author":"F Yu.","year":"2015","unstructured":"F Yu. 2015. Multi-scale context aggregation by dilated convolutions. arXiv preprint arXiv:1511.07122 (2015)."},{"key":"e_1_3_2_1_57_1","first-page":"17283","volume-title":"Lin (Eds.)","volume":"33","author":"Zaheer Manzil","year":"2020","unstructured":"Manzil Zaheer, Guru Guruganesh, Kumar Avinava Dubey, Joshua Ainslie, Chris Alberti, Santiago Ontanon, Philip Pham, Anirudh Ravula, Qifan Wang, Li Yang, and Amr Ahmed. 2020. Big Bird: Transformers for longer sequences. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 17283-17297."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20252"},{"key":"e_1_3_2_1_59_1","volume-title":"Explicit Sparse Transformer: Concentrated attention through explicit selection. CoRR","author":"Zhao Guangxiang","year":"2019","unstructured":"Guangxiang Zhao, Junyang Lin, Zhiyuan Zhang, Xuancheng Ren, Qi Su, and Xu Sun. 2019. Explicit Sparse Transformer: Concentrated attention through explicit selection. CoRR, Vol. abs\/1912.11637 (2019). arXiv:1912.11637 http:\/\/arxiv.org\/abs\/1912.11637"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.544"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"62442","author":"Zhu Lianghui","year":"2024","unstructured":"Lianghui Zhu, Bencheng Liao, Qian Zhang, Xinlong Wang, Wenyu Liu, and Xinggang Wang. 2024. Vision Mamba: Efficient visual representation learning with bidirectional state space model. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235), Ruslan Salakhutdinov, Zico Kolter, Katherine Heller, Adrian Weller, Nuria Oliver, Jonathan Scarlett, and Felix Berkenkamp (Eds.). PMLR, 62429-62442."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754825","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:45:26Z","timestamp":1765309526000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754825"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":61,"alternative-id":["10.1145\/3746027.3754825","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754825","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}