{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T06:25:07Z","timestamp":1783059907028,"version":"3.54.6"},"reference-count":261,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T00:00:00Z","timestamp":1777334400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100024370","name":"Ministero dell&apos;Istruzione dell&apos;Universit\u00e0 e della Ricerca","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100024370","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.cviu.2026.104791","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T20:26:11Z","timestamp":1777580771000},"page":"104791","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Self-attention as the backbone: A survey on Vision Transformers"],"prefix":"10.1016","volume":"269","author":[{"given":"Ahmad","family":"Waseem","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pietro","family":"Ruiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seth","family":"Nixon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andrea","family":"Lagorio","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Massimo","family":"Tistarelli","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104791_b1","doi-asserted-by":"crossref","unstructured":"Abnar,\u00a0S., Zuidema,\u00a0W., 2020. Quantifying attention flow in transformers. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics. pp. 4190\u20134197.","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"10.1016\/j.cviu.2026.104791_b2","unstructured":"Achtibat,\u00a0R., Hatefi,\u00a0S., Dreyer,\u00a0M., Jain,\u00a0A., Wiegand,\u00a0T., Lapuschkin,\u00a0S., Samek,\u00a0W., 2024. AttnLRP: Attention-aware layer-wise relevance propagation for transformers. In: Forty-First International Conference on Machine Learning."},{"key":"10.1016\/j.cviu.2026.104791_b3","first-page":"20014","article-title":"Xcit: Cross-covariance image transformers","volume":"34","author":"Ali","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b4","doi-asserted-by":"crossref","first-page":"2245","DOI":"10.1109\/TPAMI.2024.3506283","article-title":"Foundation models defining a new era in vision: a survey and outlook","volume":"47","author":"Awais","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b5","unstructured":"Bahdanau,\u00a0D., Cho,\u00a0K., Bengio,\u00a0Y., 2015. Neural machine translation by jointly learning to align and translate. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b6","series-title":"Qwen3-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b7","doi-asserted-by":"crossref","unstructured":"Bao,\u00a0F., Nie,\u00a0S., Xue,\u00a0K., Cao,\u00a0Y., Li,\u00a0C., Su,\u00a0H., Zhu,\u00a0J., 2023. All are worth words: A vit backbone for diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 22669\u201322679.","DOI":"10.1109\/CVPR52729.2023.02171"},{"key":"10.1016\/j.cviu.2026.104791_b8","doi-asserted-by":"crossref","unstructured":"BaoLong,\u00a0N., Zhang,\u00a0C., Shi,\u00a0Y., Hirakawa,\u00a0T., Yamashita,\u00a0T., Matsui,\u00a0T., Fujiyoshi,\u00a0H., 2024. Debiformer: Vision transformer with deformable agent bi-level routing attention. In: Proceedings of the Asian Conference on Computer Vision. ACCV, pp. 4455\u20134472.","DOI":"10.1007\/978-981-96-0972-7_26"},{"key":"10.1016\/j.cviu.2026.104791_b9","first-page":"22614","article-title":"Revisiting resnets: Improved training and scaling strategies","volume":"34","author":"Bello","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b10","unstructured":"Bolya,\u00a0D., Fu,\u00a0C., Dai,\u00a0X., Zhang,\u00a0P., Feichtenhofer,\u00a0C., Hoffman,\u00a0J., 2023. Token merging: Your vit but faster(tome). In: The Eleventh International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b11","doi-asserted-by":"crossref","unstructured":"Bousselham,\u00a0W., Boggust,\u00a0A., Chaybouti,\u00a0S., Strobelt,\u00a0H., Kuehne,\u00a0H., 2025. Legrad: An explainability method for vision transformers via feature formation sensitivity. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 20336\u201320345.","DOI":"10.1109\/ICCV51701.2025.01891"},{"key":"10.1016\/j.cviu.2026.104791_b12","series-title":"Dualx-vsr: Dual axial spatial x temporal transformer for real-world video super-resolution without motion compensation","author":"Cao","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b13","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.cviu.2026.104791_b14","unstructured":"Chang,\u00a0Z., Yin,\u00a0M., Wang,\u00a0Y., 2024. Coatformer: vision transformer with composite attention. In: Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. pp. 614\u2013622."},{"key":"10.1016\/j.cviu.2026.104791_b15","unstructured":"Chen,\u00a0Z., Duan,\u00a0Y., Wang,\u00a0W., He,\u00a0J., Lu,\u00a0T., Dai,\u00a0J., Qiao,\u00a0Y., 2023b. Vision transformer adapter for dense predictions. In: The Eleventh International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b16","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0C., Fan,\u00a0Q., Panda,\u00a0R., 2021. CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image Classification. In: International Conference on Computer Vision. ICCV, pp. 357\u2013366.","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"10.1016\/j.cviu.2026.104791_b17","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0X., Liu,\u00a0Z., Tang,\u00a0H., Yi,\u00a0L., Zhao,\u00a0H., Han,\u00a0S., 2023a. Sparsevit: Revisiting activation sparsity for efficient high-resolution vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2061\u20132070.","DOI":"10.1109\/CVPR52729.2023.00205"},{"key":"10.1016\/j.cviu.2026.104791_b18","unstructured":"Chen,\u00a0C., Panda,\u00a0R., Fan,\u00a0Q., 2022a. Regionvit: Regional-to-local attention for vision transformers. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b19","unstructured":"Chen,\u00a0J., Tsvetkov,\u00a0Y., Han,\u00a0X., 2026. {MADF}: Mixed autoregressive and diffusion transformers for continuous image generation. In: The Fourteenth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b20","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0Q., Wu,\u00a0Q., Wang,\u00a0J., Hu,\u00a0Q., Hu,\u00a0T., Ding,\u00a0E., Cheng,\u00a0J., Wang,\u00a0J., 2022b. Mixformer: Mixing features across windows and dimensions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 5249\u20135259.","DOI":"10.1109\/CVPR52688.2022.00518"},{"key":"10.1016\/j.cviu.2026.104791_b21","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0S., Xu,\u00a0M., Ren,\u00a0J., Cong,\u00a0Y., He,\u00a0S., Xie,\u00a0Y., Sinha,\u00a0A., Luo,\u00a0P., Xiang,\u00a0T., Perez-Rua,\u00a0J., 2024. Gentron: Diffusion transformers for image and video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6441\u20136451.","DOI":"10.1109\/CVPR52733.2024.00616"},{"key":"10.1016\/j.cviu.2026.104791_b22","series-title":"2016 Conference on Empirical Methods in Natural Language Processing","first-page":"551","article-title":"Long short-term memory-networks for machine reading","author":"Cheng","year":"2016"},{"key":"10.1016\/j.cviu.2026.104791_b23","doi-asserted-by":"crossref","unstructured":"Choi,\u00a0H., Jin,\u00a0S., Han,\u00a0K., 2023. Adversarial normalization: I can visualize everything (ice). In: Proceedings of the IEEE\/Cvf Conference on Computer Vision and Pattern Recognition. pp. 12115\u201312124.","DOI":"10.1109\/CVPR52729.2023.01166"},{"key":"10.1016\/j.cviu.2026.104791_b24","doi-asserted-by":"crossref","first-page":"2487","DOI":"10.1007\/s11263-024-02290-6","article-title":"Icev2: Interpretability, comprehensiveness, and explainability in vision transformer","volume":"133","author":"Choi","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.cviu.2026.104791_b25","first-page":"9355","article-title":"Twins: Revisiting the design of spatial attention in vision transformers","volume":"34","author":"Chu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b26","doi-asserted-by":"crossref","unstructured":"Cocchi,\u00a0F., Moratelli,\u00a0N., Caffagni,\u00a0D., Sarto,\u00a0S., Baraldi,\u00a0L., Cornia,\u00a0M., Cucchiara,\u00a0R., 2025. Llava-more: A comparative study of llms and visual backbones for enhanced visual instruction tuning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 4278\u20134288.","DOI":"10.1109\/ICCVW69036.2025.00450"},{"key":"10.1016\/j.cviu.2026.104791_b27","doi-asserted-by":"crossref","first-page":"201","DOI":"10.1038\/nrn755","article-title":"Control of goal-directed and stimulus-driven attention in the brain","volume":"3","author":"Corbetta","year":"2002","journal-title":"Nature Rev. Neurosci."},{"key":"10.1016\/j.cviu.2026.104791_b28","doi-asserted-by":"crossref","unstructured":"Cui,\u00a0Y., Jiang,\u00a0C., Wang,\u00a0L., Wu,\u00a0G., 2022. Mixformer: End-to-end tracking with iterative mixed attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13608\u201313618.","DOI":"10.1109\/CVPR52688.2022.01324"},{"key":"10.1016\/j.cviu.2026.104791_b29","unstructured":"Dao,\u00a0T., Gu,\u00a0A., Transformers are ssms: Generalized models and efficient algorithms through structured state space duality. In: Forty-First International Conference on Machine Learning."},{"key":"10.1016\/j.cviu.2026.104791_b30","series-title":"International Conference on Machine Learning","first-page":"2286","article-title":"Convit: Improving vision transformers with soft convolutional inductive biases","author":"d\u2019Ascoli","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b31","series-title":"2023 IEEE International Symposium on High-Performance Computer Architecture","first-page":"415","article-title":"Vitality: Unifying low-rank and sparse approximation for vision transformer acceleration with a linear taylor attention","author":"Dass","year":"2023"},{"key":"10.1016\/j.cviu.2026.104791_b32","unstructured":"Dehghani,\u00a0M., Gouws,\u00a0S., Vinyals,\u00a0O., Uszkoreit,\u00a0J., Kaiser,\u00a0L., 2019. Universal transformers. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b33","series-title":"2009 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"248","article-title":"Imagenet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.cviu.2026.104791_b34","doi-asserted-by":"crossref","unstructured":"Devlin,\u00a0J., Chang,\u00a0M., Lee,\u00a0K., Toutanova,\u00a0K., 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.cviu.2026.104791_b35","doi-asserted-by":"crossref","unstructured":"Di,\u00a0X., Peng,\u00a0L., Xia,\u00a0P., Li,\u00a0W., Pei,\u00a0R., Cao,\u00a0Y., Wang,\u00a0Y., Zha,\u00a0Z., 2025. Qmambabsr: Burst image super-resolution with query state space model. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 23080\u201323090.","DOI":"10.1109\/CVPR52734.2025.02149"},{"key":"10.1016\/j.cviu.2026.104791_b36","series-title":"Computer Vision\u2013ECCV 2022: 17th European Conference, October 23\u201327, 2022, Proceedings, Part XXIV","first-page":"74","article-title":"Davit: Dual attention vision transformers","author":"Ding","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b37","doi-asserted-by":"crossref","unstructured":"Dong,\u00a0X., Bao,\u00a0J., Chen,\u00a0D., Zhang,\u00a0W., Yu,\u00a0N., Yuan,\u00a0L., Chen,\u00a0D., Guo,\u00a0B., 2022. Cswin transformer: A general vision transformer backbone with cross-shaped windows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12124\u201312134.","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"10.1016\/j.cviu.2026.104791_b38","unstructured":"Dosovitskiy,\u00a0A., Beyer,\u00a0L., Kolesnikov,\u00a0A., Weissenborn,\u00a0D., Zhai,\u00a0X., Unterthiner,\u00a0T., Dehghani,\u00a0M., Minderer,\u00a0M., Heigold,\u00a0G., Gelly,\u00a0S., Uszkoreit,\u00a0J., Houlsby,\u00a0N., 2021. An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b39","doi-asserted-by":"crossref","first-page":"79024","DOI":"10.52202\/075280-3456","article-title":"Diversify your vision datasets with automatic diffusion-based augmentation","volume":"36","author":"Dunlap","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b40","doi-asserted-by":"crossref","unstructured":"Eberle,\u00a0O., Brandl,\u00a0S., Pilot,\u00a0J., S\u00f8gaard,\u00a0A., 2022. Do transformer models show similar attention patterns to task-specific human gaze?. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 4295\u20134309.","DOI":"10.18653\/v1\/2022.acl-long.296"},{"key":"10.1016\/j.cviu.2026.104791_b41","unstructured":"El-Nouby,\u00a0A., Klein,\u00a0M., Zhai,\u00a0S., Bautista,\u00a0M., Shankar,\u00a0V., Toshev,\u00a0A., Susskind,\u00a0J., Joulin,\u00a0A., 2024. Scalable pre-training of large autoregressive image models. In: Proceedings of the 41st International Conference on Machine Learning. pp. 12371\u201312384."},{"key":"10.1016\/j.cviu.2026.104791_b42","doi-asserted-by":"crossref","unstructured":"Englebert,\u00a0A., Stassin,\u00a0S., Nanfack,\u00a0G., Mahmoudi,\u00a0S., Siebert,\u00a0X., Cornu,\u00a0O., De\u00a0Vleeschouwer,\u00a0C., 2023. Explaining through transformer input sampling. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 806\u2013815.","DOI":"10.1109\/ICCVW60793.2023.00088"},{"key":"10.1016\/j.cviu.2026.104791_b43","doi-asserted-by":"crossref","unstructured":"Fan,\u00a0Q., Huang,\u00a0H., He,\u00a0R., 2025. Breaking the low-rank dilemma of linear attention. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 25271\u201325280.","DOI":"10.1109\/CVPR52734.2025.02353"},{"key":"10.1016\/j.cviu.2026.104791_b44","doi-asserted-by":"crossref","unstructured":"Fan,\u00a0H., Xiong,\u00a0B., Mangalam,\u00a0K., Li,\u00a0Y., Yan,\u00a0Z., Malik,\u00a0J., Feichtenhofer,\u00a0C., 2021. Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6824\u20136835.","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"10.1016\/j.cviu.2026.104791_b45","doi-asserted-by":"crossref","unstructured":"Fang,\u00a0J., Xie,\u00a0L., Wang,\u00a0X., Zhang,\u00a0X., Liu,\u00a0W., Tian,\u00a0Q., 2022. Msg-transformer: Exchanging local spatial information by manipulating messenger tokens. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12063\u201312072.","DOI":"10.1109\/CVPR52688.2022.01175"},{"key":"10.1016\/j.cviu.2026.104791_b46","doi-asserted-by":"crossref","first-page":"92","DOI":"10.3390\/computers13040092","article-title":"The explainability of transformers: Current status and directions","volume":"13","author":"Fantozzi","year":"2024","journal-title":"Computers"},{"key":"10.1016\/j.cviu.2026.104791_b47","series-title":"European Conference on Computer Vision","first-page":"396","article-title":"Adaptive token sampling for efficient vision transformers","author":"Fayyaz","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b48","doi-asserted-by":"crossref","unstructured":"Fini,\u00a0E., Shukor,\u00a0M., Li,\u00a0X., Dufter,\u00a0P., Klein,\u00a0M., Haldimann,\u00a0D., Aitharaju,\u00a0S., da\u00a0Costa,\u00a0V., B\u00e9thune,\u00a0L., Gan,\u00a0Z., et al., 2025. Multimodal autoregressive pre-training of large vision encoders. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 9641\u20139654.","DOI":"10.1109\/CVPR52734.2025.00901"},{"key":"10.1016\/j.cviu.2026.104791_b49","unstructured":"Gandelsman,\u00a0Y., Efros,\u00a0A., Steinhardt,\u00a0J., 2024. Interpreting CLIP\u2019s image representation via text-based decomposition. In: The Twelfth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b50","doi-asserted-by":"crossref","unstructured":"Gao,\u00a0S., Zhou,\u00a0P., Cheng,\u00a0M., Yan,\u00a0S., 2023. Masked diffusion transformer is a strong image synthesizer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 23164\u201323173.","DOI":"10.1109\/ICCV51070.2023.02117"},{"key":"10.1016\/j.cviu.2026.104791_b51","doi-asserted-by":"crossref","unstructured":"Graham,\u00a0B., El-Nouby,\u00a0A., Touvron,\u00a0H., Stock,\u00a0P., Joulin,\u00a0A., J\u00e9gou,\u00a0H., Douze,\u00a0M., 2021. Levit: A vision transformer in convnet\u2019s clothing for faster inference. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 12259\u201312269.","DOI":"10.1109\/ICCV48922.2021.01204"},{"key":"10.1016\/j.cviu.2026.104791_b52","doi-asserted-by":"crossref","unstructured":"Grainger,\u00a0R., Paniagua,\u00a0T., Song,\u00a0X., Cuntoor,\u00a0N., Lee,\u00a0M., Wu,\u00a0T., 2023. Paca-vit: Learning patch-to-cluster attention in vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 18568\u201318578.","DOI":"10.1109\/CVPR52729.2023.01781"},{"key":"10.1016\/j.cviu.2026.104791_b53","series-title":"Adaptive computation time for recurrent neural networks","author":"Graves","year":"2016"},{"key":"10.1016\/j.cviu.2026.104791_b54","article-title":"The neural dynamics underlying prioritisation of task-relevant information","author":"Grootswagers","year":"2023","journal-title":"Neurons, Behav. Data Anal. Theory"},{"key":"10.1016\/j.cviu.2026.104791_b55","unstructured":"Gu,\u00a0A., Dao,\u00a0T., 2024. Mamba: Linear-time sequence modeling with selective state spaces. In: First Conference on Language Modeling."},{"key":"10.1016\/j.cviu.2026.104791_b56","doi-asserted-by":"crossref","unstructured":"Gu,\u00a0J., Kwon,\u00a0H., Wang,\u00a0D., Ye,\u00a0W., Li,\u00a0M., Chen,\u00a0Y., Lai,\u00a0L., Chandra,\u00a0V., Pan,\u00a0D., 2022. Multi-scale high-resolution vision transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 12094\u201312103.","DOI":"10.1109\/CVPR52688.2022.01178"},{"key":"10.1016\/j.cviu.2026.104791_b57","doi-asserted-by":"crossref","unstructured":"Guo,\u00a0J., Han,\u00a0K., Wu,\u00a0H., Tang,\u00a0Y., Chen,\u00a0X., Wang,\u00a0Y., Xu,\u00a0C., 2022a. Cmt: Convolutional neural networks meet vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12175\u201312185.","DOI":"10.1109\/CVPR52688.2022.01186"},{"key":"10.1016\/j.cviu.2026.104791_b58","first-page":"5436","article-title":"Beyond self-attention: External attention using two linear layers for visual tasks","volume":"45","author":"Guo","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b59","first-page":"5436","article-title":"Beyond self-attention: External attention using two linear layers for visual tasks","volume":"45","author":"Guo","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b60","doi-asserted-by":"crossref","first-page":"733","DOI":"10.1007\/s41095-023-0364-2","article-title":"Visual attention network","volume":"9","author":"Guo","year":"2023","journal-title":"Comput. Vis. Media"},{"key":"10.1016\/j.cviu.2026.104791_b61","doi-asserted-by":"crossref","unstructured":"Guo,\u00a0R., Niu,\u00a0D., Qu,\u00a0L., Li,\u00a0Z., 2021. Sotr: Segmenting objects with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 7157\u20137166.","DOI":"10.1109\/ICCV48922.2021.00707"},{"key":"10.1016\/j.cviu.2026.104791_b62","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.131431","article-title":"All-in-one weather image restoration: Multi-modal attention and bi-directional mamba fusion","author":"Guo","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104791_b63","doi-asserted-by":"crossref","unstructured":"Han,\u00a0D., Pan,\u00a0X., Han,\u00a0Y., Song,\u00a0S., Huang,\u00a0G., 2023. Flatten transformer: Vision transformer using focused linear attention. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 5961\u20135971.","DOI":"10.1109\/ICCV51070.2023.00548"},{"key":"10.1016\/j.cviu.2026.104791_b64","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A survey on vision transformer","volume":"45","author":"Han","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b65","doi-asserted-by":"crossref","first-page":"127181","DOI":"10.52202\/079017-4039","article-title":"Demystify mamba in vision: A linear attention perspective","volume":"37","author":"Han","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b66","unstructured":"Han,\u00a0K., Xiao,\u00a0A., Wu,\u00a0E., Guo,\u00a0J., Xu,\u00a0C., Wang,\u00a0Y., 2021. Transformer in transformer. In: Advances in Neural Information Processing Systems. pp. 15908\u201315919."},{"key":"10.1016\/j.cviu.2026.104791_b67","doi-asserted-by":"crossref","unstructured":"Hassani,\u00a0A., Walton,\u00a0S., Li,\u00a0J., Li,\u00a0S., Shi,\u00a0H., 2023. Neighborhood attention transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6185\u20136194.","DOI":"10.1109\/CVPR52729.2023.00599"},{"key":"10.1016\/j.cviu.2026.104791_b68","unstructured":"Hatamizadeh,\u00a0A., Heinrich,\u00a0G., Yin,\u00a0H., Tao,\u00a0A., Alvarez,\u00a0J., Kautz,\u00a0J., Molchanov,\u00a0P., 2024a. Fastervit: Fast vision transformers with hierarchical attention. In: The Twelfth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b69","doi-asserted-by":"crossref","unstructured":"Hatamizadeh,\u00a0A., Kautz,\u00a0J., 2025. Mambavision: A hybrid mamba-transformer vision backbone. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 25261\u201325270.","DOI":"10.1109\/CVPR52734.2025.02352"},{"key":"10.1016\/j.cviu.2026.104791_b70","series-title":"European Conference on Computer Vision","first-page":"37","article-title":"Diffit: Diffusion vision transformers for image generation","author":"Hatamizadeh","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b71","series-title":"International Conference on Machine Learning","first-page":"12633","article-title":"Global context vision transformers","author":"Hatamizadeh","year":"2023"},{"key":"10.1016\/j.cviu.2026.104791_b72","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Chen,\u00a0X., Xie,\u00a0S., Li,\u00a0Y., Doll\u00e1r,\u00a0P., Girshick,\u00a0R., 2022. Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10.1016\/j.cviu.2026.104791_b73","unstructured":"He,\u00a0P., Liu,\u00a0X., Gao,\u00a0J., Chen,\u00a0W., 2021. Deberta: Decoding-enhanced bert with disentangled attention. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b74","doi-asserted-by":"crossref","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","article-title":"Spatial pyramid pooling in deep convolutional networks for visual recognition","volume":"37","author":"He","year":"2015","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b75","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Zhang,\u00a0X., Ren,\u00a0S., Sun,\u00a0J., 2016. Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.cviu.2026.104791_b76","doi-asserted-by":"crossref","first-page":"94097","DOI":"10.52202\/079017-2985","article-title":"Perceiving longer sequences with bi-directional cross-attention transformers","volume":"37","author":"Hiller","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b77","series-title":"Axial attention in multidimensional transformers","author":"Ho","year":"2019"},{"key":"10.1016\/j.cviu.2026.104791_b78","series-title":"Shuffle transformer: Rethinking spatial shuffle for vision transformer","author":"Huang","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b79","series-title":"Lightvit: Towards light-weight convolution-free vision transformers","author":"Huang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b80","series-title":"International Conference on Machine Learning","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","author":"Jaegle","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b81","unstructured":"Jain,\u00a0S., Wallace,\u00a0B., 2019. Attention is not explanation. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). pp. 3543\u20133556."},{"key":"10.1016\/j.cviu.2026.104791_b82","doi-asserted-by":"crossref","unstructured":"Jain,\u00a0J., Yang,\u00a0J., Shi,\u00a0H., 2024. Vcoder: Versatile vision encoders for multimodal large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 27992\u201328002.","DOI":"10.1109\/CVPR52733.2024.02644"},{"key":"10.1016\/j.cviu.2026.104791_b83","first-page":"14745","article-title":"Transgan: Two pure transformers can make one strong gan, and that can scale up","volume":"34","author":"Jiang","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b84","first-page":"18590","article-title":"All tokens matter: Token labeling for training better vision transformers","volume":"34","author":"Jiang","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b85","doi-asserted-by":"crossref","first-page":"8906","DOI":"10.1109\/TMM.2023.3243616","article-title":"Dilateformer: Multi-scale dilated transformer for visual recognition","volume":"25","author":"Jiao","year":"2023","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.cviu.2026.104791_b86","series-title":"2025 IEEE International Conference on Image Processing","first-page":"582","article-title":"Gmar: gradient-driven multi-head attention rollout for vision transformer interpretability","author":"Jo","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b87","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103184","article-title":"Explainability and vision foundation models: A survey","volume":"122","author":"Kazmierczak","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.cviu.2026.104791_b88","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3505244","article-title":"Transformers in vision: A survey","volume":"54","author":"Khan","year":"2022","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.cviu.2026.104791_b89","doi-asserted-by":"crossref","first-page":"2917","DOI":"10.1007\/s10462-023-10595-0","article-title":"A survey of the vision transformers and their cnn-transformer based variants","volume":"56","author":"Khan","year":"2023","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.cviu.2026.104791_b90","doi-asserted-by":"crossref","unstructured":"Kim,\u00a0Y., Anagnostidis,\u00a0S., Du,\u00a0Y., Sch\u00f6nfeld,\u00a0E., Kohler,\u00a0J., Georgopoulos,\u00a0M., Pumarola,\u00a0A., Thabet,\u00a0A., Sanakoyeu,\u00a0A., 2025. Autoregressive distillation of diffusion transformers. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 15745\u201315756.","DOI":"10.1109\/CVPR52734.2025.01468"},{"key":"10.1016\/j.cviu.2026.104791_b91","doi-asserted-by":"crossref","unstructured":"Kirillov,\u00a0A., Mintun,\u00a0E., Ravi,\u00a0N., Mao,\u00a0H., Rolland,\u00a0C., Gustafson,\u00a0L., Xiao,\u00a0T., Whitehead,\u00a0S., Berg,\u00a0A., Lo,\u00a0W., et al., 2023. Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 4015\u20134026.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"10.1016\/j.cviu.2026.104791_b92","doi-asserted-by":"crossref","unstructured":"Kockwelp,\u00a0J., Beckmann,\u00a0D., Risse,\u00a0B., 2025. Human gaze improves vision transformers by token masking. In: Proceedings of the Winter Conference on Applications of Computer Vision. pp. 396\u2013405.","DOI":"10.1109\/WACVW65960.2025.00046"},{"key":"10.1016\/j.cviu.2026.104791_b93","series-title":"European Conference on Computer Vision","first-page":"620","article-title":"Spvit: Enabling faster vision transformers via latency-aware soft token pruning","author":"Kong","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b94","doi-asserted-by":"crossref","unstructured":"Koohpayegani,\u00a0S., Pirsiavash,\u00a0H., 2024. Sima: Simple softmax-free attention for vision transformers. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2607\u20132617.","DOI":"10.1109\/WACV57701.2024.00259"},{"key":"10.1016\/j.cviu.2026.104791_b95","series-title":"2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition","first-page":"2169","article-title":"Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories","author":"Lazebnik","year":"2006"},{"key":"10.1016\/j.cviu.2026.104791_b96","unstructured":"Lee,\u00a0K., Chang,\u00a0H., Jiang,\u00a0L., Zhang,\u00a0H., Tu,\u00a0Z., Liu,\u00a0C., 2022. ViTGAN: Training GANs with vision transformers. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b97","doi-asserted-by":"crossref","unstructured":"Lee,\u00a0S., Choi,\u00a0J., Kim,\u00a0H., 2024. Multi-criteria token fusion with one-step-ahead attention for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 15741\u201315750.","DOI":"10.1109\/CVPR52733.2024.01490"},{"key":"10.1016\/j.cviu.2026.104791_b98","doi-asserted-by":"crossref","unstructured":"Lee,\u00a0S., Choi,\u00a0J., Kim,\u00a0H., 2025. Efficientvim: Efficient vision mamba with hidden state mixer based state space duality. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 14923\u201314933.","DOI":"10.1109\/CVPR52734.2025.01390"},{"key":"10.1016\/j.cviu.2026.104791_b99","series-title":"Vision transformer for small-size datasets","author":"Lee","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b100","doi-asserted-by":"crossref","unstructured":"Leem,\u00a0S., Seo,\u00a0H., 2024. Attention guided cam: visual explanations of vision transformer guided by self-attention. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 2956\u20132964.","DOI":"10.1609\/aaai.v38i4.28077"},{"key":"10.1016\/j.cviu.2026.104791_b101","doi-asserted-by":"crossref","unstructured":"Lei,\u00a0W., Ge,\u00a0Y., Yi,\u00a0K., Zhang,\u00a0J., Gao,\u00a0D., Sun,\u00a0D., Ge,\u00a0Y., Shan,\u00a0Y., Shou,\u00a0M., 2024. Vit-lens: Towards omni-modal representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 26647\u201326657.","DOI":"10.1109\/CVPR52733.2024.02516"},{"key":"10.1016\/j.cviu.2026.104791_b102","doi-asserted-by":"crossref","first-page":"12772","DOI":"10.1109\/TNNLS.2023.3264730","article-title":"Bvit: Broad attention-based vision transformer","volume":"35","author":"Li","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b103","doi-asserted-by":"crossref","unstructured":"Li,\u00a0B., Hu,\u00a0Y., Nie,\u00a0X., Han,\u00a0C., Jiang,\u00a0X., Guo,\u00a0T., Liu,\u00a0L., 2023a. Dropkey for vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 22700\u201322709.","DOI":"10.1109\/CVPR52729.2023.02174"},{"key":"10.1016\/j.cviu.2026.104791_b104","doi-asserted-by":"crossref","unstructured":"Li,\u00a0Y., Hu,\u00a0J., Wen,\u00a0Y., Evangelidis,\u00a0G., Salahi,\u00a0K., Wang,\u00a0Y., Tulyakov,\u00a0S., Ren,\u00a0J., 2023e. Efficientformerv2: Rethinking vision transformers for mobilenet size and speed. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 16889\u201316900.","DOI":"10.1109\/ICCV51070.2023.01549"},{"key":"10.1016\/j.cviu.2026.104791_b105","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.cviu.2026.104791_b106","series-title":"European Conference on Computer Vision","first-page":"237","article-title":"Videomamba: State space model for efficient video understanding","author":"Li","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b107","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b108","doi-asserted-by":"crossref","unstructured":"Li,\u00a0Y., Liao,\u00a0B., Liu,\u00a0W., Wang,\u00a0X., 2025. Matvlm: Hybrid mamba-transformer for efficient vision-language modeling. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV.","DOI":"10.1109\/ICCV51701.2025.01941"},{"key":"10.1016\/j.cviu.2026.104791_b109","doi-asserted-by":"crossref","first-page":"56424","DOI":"10.52202\/079017-1797","article-title":"Autoregressive image generation without vector quantization","volume":"37","author":"Li","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b110","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106653","article-title":"Diagswin: A multi-scale vision transformer with diagonal-shaped windows for object detection and segmentation","volume":"180","author":"Li","year":"2024","journal-title":"Neural Netw."},{"key":"10.1016\/j.cviu.2026.104791_b111","doi-asserted-by":"crossref","first-page":"12581","DOI":"10.1109\/TPAMI.2023.3282631","article-title":"Uniformer: Unifying convolution and self-attention for visual recognition","volume":"45","author":"Li","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b112","doi-asserted-by":"crossref","unstructured":"Li,\u00a0Y., Wu,\u00a0C., Fan,\u00a0H., Mangalam,\u00a0K., Xiong,\u00a0B., Malik,\u00a0J., Feichtenhofer,\u00a0C., 2022c. Mvitv2: Improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4804\u20134814.","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"10.1016\/j.cviu.2026.104791_b113","series-title":"Next-vit: Next generation vision transformer for efficient deployment in realistic industrial scenarios","author":"Li","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b114","series-title":"Local-to-global self-attention in vision transformers","author":"Li","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b115","doi-asserted-by":"crossref","first-page":"1489","DOI":"10.1109\/TPAMI.2022.3164083","article-title":"Contextual transformer networks for visual recognition","volume":"45","author":"Li","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b116","doi-asserted-by":"crossref","first-page":"12934","DOI":"10.52202\/068431-0940","article-title":"Efficientformer: Vision transformers at mobilenet speed","volume":"35","author":"Li","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b117","doi-asserted-by":"crossref","unstructured":"Li,\u00a0G., Zhu,\u00a0L., Liu,\u00a0P., Yang,\u00a0Y., 2019. Entangled transformer for image captioning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 8928\u20138937.","DOI":"10.1109\/ICCV.2019.00902"},{"key":"10.1016\/j.cviu.2026.104791_b118","unstructured":"Liang,\u00a0Y., Ge,\u00a0C., Tong,\u00a0Z., Song,\u00a0Y., Wang,\u00a0J., Xie,\u00a0P., 2022. Not all patches are what you need: Expediting vision transformers via token reorganizations. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b119","series-title":"2022 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Cat: Cross attention in vision transformer","author":"Lin","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b120","doi-asserted-by":"crossref","unstructured":"Lin,\u00a0T., Doll\u00e1r,\u00a0P., Girshick,\u00a0R., He,\u00a0K., Hariharan,\u00a0B., Belongie,\u00a0S., 2017. Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 2117\u20132125.","DOI":"10.1109\/CVPR.2017.106"},{"key":"10.1016\/j.cviu.2026.104791_b121","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0H., Li,\u00a0C., Li,\u00a0Y., Lee,\u00a0Y., 2024a. Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.cviu.2026.104791_b122","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b123","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Z., Lin,\u00a0Y., Cao,\u00a0Y., Hu,\u00a0H., Wei,\u00a0Y., Zhang,\u00a0Z., Lin,\u00a0S., Guo,\u00a0B., 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.cviu.2026.104791_b124","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0X., Peng,\u00a0H., Zheng,\u00a0N., Yang,\u00a0Y., Hu,\u00a0H., Yuan,\u00a0Y., 2023b. Efficientvit: Memory efficient vision transformer with cascaded group attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 14420\u201314430.","DOI":"10.1109\/CVPR52729.2023.01386"},{"key":"10.1016\/j.cviu.2026.104791_b125","doi-asserted-by":"crossref","first-page":"103031","DOI":"10.52202\/079017-3273","article-title":"Vmamba: Visual state space model","volume":"37","author":"Liu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b126","series-title":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence, IJCAI-22","first-page":"1187","article-title":"Dynamic group transformer: A general vision transformer backbone with dynamic group attention","author":"Liu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b127","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Y., Yi,\u00a0L., 2025. Map: Unleashing hybrid mamba-transformer vision backbone\u2019s potential with masked autoregressive pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 9676\u20139685.","DOI":"10.1109\/CVPR52734.2025.00904"},{"key":"10.1016\/j.cviu.2026.104791_b128","doi-asserted-by":"crossref","first-page":"7478","DOI":"10.1109\/TNNLS.2022.3227717","article-title":"A survey of visual transformers","volume":"35","author":"Liu","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b129","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0L., Zhang,\u00a0M., Yin,\u00a0J., Liu,\u00a0T., Ji,\u00a0W., Piao,\u00a0Y., Lu,\u00a0H., 2025. Defmamba: Deformable visual state space model. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 8838\u20138847.","DOI":"10.1109\/CVPR52734.2025.00826"},{"key":"10.1016\/j.cviu.2026.104791_b130","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1109\/TIT.1982.1056489","article-title":"Least squares quantization in pcm","volume":"28","author":"Lloyd","year":"1982","journal-title":"IEEE Trans. Inform. Theory"},{"key":"10.1016\/j.cviu.2026.104791_b131","doi-asserted-by":"crossref","unstructured":"Long,\u00a0S., Zhao,\u00a0Z., Pi,\u00a0J., Wang,\u00a0S., Wang,\u00a0J., 2023. Beyond attentive tokens: Incorporating token importance and diversity for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 10334\u201310343.","DOI":"10.1109\/CVPR52729.2023.00996"},{"key":"10.1016\/j.cviu.2026.104791_b132","series-title":"A2mamba: Attention-augmented state space models for visual recognition","author":"Lou","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b133","doi-asserted-by":"crossref","unstructured":"Lu,\u00a0Z., Li,\u00a0J., Liu,\u00a0H., Huang,\u00a0C., Zhang,\u00a0L., Zeng,\u00a0T., 2022b. Transformer for single image super-resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops. pp. 457\u2013466.","DOI":"10.1109\/CVPRW56347.2022.00061"},{"key":"10.1016\/j.cviu.2026.104791_b134","series-title":"International Conference on Machine Learning","first-page":"33160","article-title":"Fit: Flexible vision transformer for diffusion model","author":"Lu","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b135","first-page":"21297","article-title":"Soft: Softmax-free transformer with linear complexity","volume":"34","author":"Lu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b136","doi-asserted-by":"crossref","first-page":"5775","DOI":"10.52202\/068431-0418","article-title":"Dpm-solver: A fast ode solver for diffusion probabilistic model sampling in around 10 steps","volume":"35","author":"Lu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b137","series-title":"European Conference on Computer Vision","first-page":"23","article-title":"Sit: Exploring flow and diffusion-based generative models with scalable interpolant transformers","author":"Ma","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b138","unstructured":"Maddison,\u00a0C., Mnih,\u00a0A., Teh,\u00a0Y., 2017. The concrete distribution: A continuous relaxation of discrete random variables. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b139","doi-asserted-by":"crossref","unstructured":"Mao,\u00a0X., Qi,\u00a0G., Chen,\u00a0Y., Li,\u00a0X., Duan,\u00a0R., Ye,\u00a0S., He,\u00a0Y., Xue,\u00a0H., 2022. Towards robust vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12042\u201312051.","DOI":"10.1109\/CVPR52688.2022.01173"},{"key":"10.1016\/j.cviu.2026.104791_b140","doi-asserted-by":"crossref","first-page":"24385","DOI":"10.1007\/s00521-025-11591-x","article-title":"Multiplex network-based representation of vision transformers for visual explainability","volume":"37","author":"Marchetti","year":"2025","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.cviu.2026.104791_b141","series-title":"Adavit: Adaptive vision transformers for efficient image recognition","first-page":"12309","author":"Meng","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b142","doi-asserted-by":"crossref","first-page":"48","DOI":"10.1016\/j.neucom.2021.03.091","article-title":"A review on the attention mechanism of deep learning","volume":"452","author":"Niu","year":"2021","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104791_b143","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.cviu.2026.104791_b144","series-title":"European Conference on Computer Vision","first-page":"294","article-title":"Edgevits: Competing light-weight cnns on mobile devices with vision transformers","author":"Pan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b145","doi-asserted-by":"crossref","first-page":"14541","DOI":"10.52202\/068431-1057","article-title":"Fast vision transformers with hilo attention","volume":"35","author":"Pan","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b146","doi-asserted-by":"crossref","unstructured":"Pan,\u00a0X., Ge,\u00a0C., Lu,\u00a0R., Song,\u00a0S., Chen,\u00a0G., Huang,\u00a0Z., Huang,\u00a0G., 2022b. On the integration of self-attention and convolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 815\u2013825.","DOI":"10.1109\/CVPR52688.2022.00089"},{"key":"10.1016\/j.cviu.2026.104791_b147","doi-asserted-by":"crossref","unstructured":"Pan,\u00a0X., Ye,\u00a0T., Xia,\u00a0Z., Song,\u00a0S., Huang,\u00a0G., 2023. Slide-transformer: Hierarchical vision transformer with local self-attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 2082\u20132091.","DOI":"10.1109\/CVPR52729.2023.00207"},{"key":"10.1016\/j.cviu.2026.104791_b148","series-title":"International Conference on Machine Learning","first-page":"4055","article-title":"Image transformer","author":"Parmar","year":"2018"},{"key":"10.1016\/j.cviu.2026.104791_b149","doi-asserted-by":"crossref","unstructured":"Peebles,\u00a0W., Xie,\u00a0S., 2023. Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 4195\u20134205.","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"10.1016\/j.cviu.2026.104791_b150","series-title":"Mathematical Proceedings of the Cambridge Philosophical Society","first-page":"406","article-title":"A generalized inverse for matrices","author":"Penrose","year":"1955"},{"key":"10.1016\/j.cviu.2026.104791_b151","series-title":"NeurIPS Efficient Natural Language and Speech Processing Workshop","first-page":"102","article-title":"Vl-mamba: Exploring state space models for multimodal learning","author":"Qiao","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b152","unstructured":"Qin,\u00a0Z., Sun,\u00a0W., Deng,\u00a0H., Li,\u00a0D., Wei,\u00a0Y., Lv,\u00a0B., Yan,\u00a0J., Kong,\u00a0L., Zhong,\u00a0Y., 2022. cosformer: Rethinking softmax in attention. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b153","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b154","series-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"10.1016\/j.cviu.2026.104791_b155","first-page":"13937","article-title":"Dynamicvit: Efficient vision transformers with dynamic token sparsification","volume":"34","author":"Rao","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b156","doi-asserted-by":"crossref","unstructured":"Ren,\u00a0W., Ma,\u00a0W., Yang,\u00a0H., Wei,\u00a0C., Zhang,\u00a0G., Chen,\u00a0W., 2025. Vamba: Understanding hour-long videos with hybrid mamba-transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 21197\u201321208.","DOI":"10.1109\/ICCV51701.2025.01969"},{"key":"10.1016\/j.cviu.2026.104791_b157","doi-asserted-by":"crossref","unstructured":"Ren,\u00a0S., Zhou,\u00a0D., He,\u00a0S., Feng,\u00a0J., Wang,\u00a0X., 2022. Shunted self-attention via multi-scale token aggregation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10853\u201310862.","DOI":"10.1109\/CVPR52688.2022.01058"},{"key":"10.1016\/j.cviu.2026.104791_b158","doi-asserted-by":"crossref","first-page":"17","DOI":"10.1080\/135062800394667","article-title":"The dynamic representation of scenes","volume":"7","author":"Rensink","year":"2000","journal-title":"Vis. Cogn."},{"key":"10.1016\/j.cviu.2026.104791_b159","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"234","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"Ronneberger","year":"2015"},{"key":"10.1016\/j.cviu.2026.104791_b160","series-title":"International Conference on Artificial Intelligence and Statistics","first-page":"3515","article-title":"Sinkformers: Transformers with doubly stochastic attention","author":"Sander","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b161","doi-asserted-by":"crossref","unstructured":"Sandler,\u00a0M., Howard,\u00a0A., Zhu,\u00a0M., Zhmoginov,\u00a0A., Chen,\u00a0L., 2018. Mobilenetv2: Inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4510\u20134520.","DOI":"10.1109\/CVPR.2018.00474"},{"key":"10.1016\/j.cviu.2026.104791_b162","series-title":"European Conference on Computer Vision","first-page":"727","article-title":"Sliced recursive transformer","author":"Shen","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b163","doi-asserted-by":"crossref","unstructured":"Shi,\u00a0Y., Li,\u00a0M., Dong,\u00a0M., Xu,\u00a0C., 2025. Vssd: Vision mamba with non-causal state space duality. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10819\u201310829.","DOI":"10.1109\/ICCV51701.2025.01007"},{"key":"10.1016\/j.cviu.2026.104791_b164","first-page":"19899","article-title":"Adder attention for vision transformer","volume":"34","author":"Shu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b165","doi-asserted-by":"crossref","unstructured":"Shukor,\u00a0M., Fini,\u00a0E., da\u00a0Costa,\u00a0V., Cord,\u00a0M., Susskind,\u00a0J., El-Nouby,\u00a0A., 2025. Scaling laws for native multimodal models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 12\u201323.","DOI":"10.1109\/ICCV51701.2025.00009"},{"key":"10.1016\/j.cviu.2026.104791_b166","doi-asserted-by":"crossref","first-page":"23495","DOI":"10.52202\/068431-1707","article-title":"Inception transformer","volume":"35","author":"Si","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b167","doi-asserted-by":"crossref","first-page":"876","DOI":"10.1214\/aoms\/1177703591","article-title":"A relationship between arbitrary positive matrices and doubly stochastic matrices","volume":"35","author":"Sinkhorn","year":"1964","journal-title":"Ann. Math. Stat."},{"key":"10.1016\/j.cviu.2026.104791_b168","series-title":"Autoregressive model beats diffusion: Llama for scalable image generation","author":"Sun","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b169","doi-asserted-by":"crossref","unstructured":"Sun,\u00a0Y., Ochiai,\u00a0H., Wu,\u00a0Z., Lin,\u00a0S., Kanai,\u00a0R., 2025. Associative transformer. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 4518\u20134527.","DOI":"10.1109\/CVPR52734.2025.00426"},{"key":"10.1016\/j.cviu.2026.104791_b170","doi-asserted-by":"crossref","first-page":"12635","DOI":"10.1109\/TPAMI.2023.3285569","article-title":"Vicinity vision transformer","volume":"45","author":"Sun","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b171","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127828","article-title":"Maformer: A transformer network with multi-scale attention fusion for visual recognition","volume":"595","author":"Sun","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104791_b172","doi-asserted-by":"crossref","unstructured":"Szegedy,\u00a0C., Liu,\u00a0W., Jia,\u00a0Y., Sermanet,\u00a0P., Reed,\u00a0S., Anguelov,\u00a0D., Erhan,\u00a0D., Vanhoucke,\u00a0V., Rabinovich,\u00a0A., 2015. Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"10.1016\/j.cviu.2026.104791_b173","series-title":"International Conference on Machine Learning","first-page":"10096","article-title":"Efficientnetv2: Smaller models and faster training","author":"Tan","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b174","doi-asserted-by":"crossref","unstructured":"Tang,\u00a0Y., Han,\u00a0K., Wang,\u00a0Y., Xu,\u00a0C., Guo,\u00a0J., Xu,\u00a0C., Tao,\u00a0D., 2022b. Patch slimming for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12165\u201312174.","DOI":"10.1109\/CVPR52688.2022.01185"},{"key":"10.1016\/j.cviu.2026.104791_b175","unstructured":"Tang,\u00a0H., Wu,\u00a0Y., Yang,\u00a0S., Xie,\u00a0E., Chen,\u00a0J., Chen,\u00a0J., Zhang,\u00a0Z., Cai,\u00a0H., Lu,\u00a0Y., Han,\u00a0S., 2025. HART: Efficient visual generation with hybrid autoregressive transformer. In: The Thirteenth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b176","unstructured":"Tang,\u00a0S., Zhang,\u00a0J., Zhu,\u00a0S., Tan,\u00a0P., 2022a. Quadtree attention for vision transformers. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b177","series-title":"International Conference on Machine Learning","first-page":"10183","article-title":"Synthesizer: Rethinking self-attention for transformer models","author":"Tay","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b178","series-title":"Gemma 3 technical report","author":"Team","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b179","doi-asserted-by":"crossref","first-page":"84839","DOI":"10.52202\/079017-2694","article-title":"Visual autoregressive modeling: Scalable image generation via next-scale prediction","volume":"37","author":"Tian","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b180","series-title":"International Conference on Machine Learning","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b181","doi-asserted-by":"crossref","unstructured":"Touvron,\u00a0H., Cord,\u00a0M., Sablayrolles,\u00a0A., Synnaeve,\u00a0G., J\u00e9gou,\u00a0H., 2021b. Going deeper with image transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 32\u201342.","DOI":"10.1109\/ICCV48922.2021.00010"},{"key":"10.1016\/j.cviu.2026.104791_b182","series-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.cviu.2026.104791_b183","series-title":"Siglip 2: Multilingual vision-language encoders with improved semantic understanding, localization, and dense features","author":"Tschannen","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b184","series-title":"European Conference on Computer Vision","first-page":"459","article-title":"Maxvit: Multi-axis vision transformer","author":"Tu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b185","series-title":"Contrast: A hybrid architecture of transformers and state space models for low-level vision","author":"Urumbekov","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b186","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b187","series-title":"European Conference on Computer Vision","first-page":"298","article-title":"Efficient vision transformers with partial attention","author":"Vo","year":"2024"},{"key":"10.1016\/j.cviu.2026.104791_b188","series-title":"57th Annual Meeting of the Association for Computational Linguistics, ACL 2019","article-title":"Analyzing multi-head self-attention: Specialized heads do the heavy lifting, the rest can be pruned","author":"Voita","year":"2019"},{"key":"10.1016\/j.cviu.2026.104791_b189","doi-asserted-by":"crossref","first-page":"3123","DOI":"10.1109\/TPAMI.2023.3341806","article-title":"Crossformer++: A versatile vision transformer hinging on cross-scale attention","volume":"46","author":"Wang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b190","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Z., Cun,\u00a0X., Bao,\u00a0J., Zhou,\u00a0W., Liu,\u00a0J., Li,\u00a0H., 2022d. Uformer: A general u-shaped transformer for image restoration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 17683\u201317693.","DOI":"10.1109\/CVPR52688.2022.01716"},{"key":"10.1016\/j.cviu.2026.104791_b191","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Y., Lin,\u00a0Z., Teng,\u00a0Y., Zhu,\u00a0Y., Ren,\u00a0S., Feng,\u00a0J., Liu,\u00a0X., 2025b. Bridging continuous and discrete tokens for autoregressive visual generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 18596\u201318605.","DOI":"10.1109\/ICCV51701.2025.01728"},{"key":"10.1016\/j.cviu.2026.104791_b192","doi-asserted-by":"crossref","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","article-title":"Deep high-resolution representation learning for visual recognition","volume":"43","author":"Wang","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b193","series-title":"European Conference on Computer Vision","first-page":"285","article-title":"Kvt: k-nn attention for boosting vision transformers","author":"Wang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b194","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0W., Xie,\u00a0E., Li,\u00a0X., Fan,\u00a0D., Song,\u00a0K., Liang,\u00a0D., Lu,\u00a0T., Luo,\u00a0P., Shao,\u00a0L., 2021. Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 568\u2013578.","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"10.1016\/j.cviu.2026.104791_b195","article-title":"A dynamic hybrid network with attention and mamba for image captioning","author":"Wang","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104791_b196","doi-asserted-by":"crossref","first-page":"8176","DOI":"10.1109\/TPAMI.2023.3236725","article-title":"Convolution-enhanced evolving attention networks","volume":"45","author":"Wang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b197","unstructured":"Wang,\u00a0W., Yao,\u00a0L., Chen,\u00a0L., Lin,\u00a0B., Cai,\u00a0D., He,\u00a0X., Liu,\u00a0W., 2022c. Crossformer: A versatile vision transformer hinging on cross-scale attention. In: International Conference on Learning Representations, ICLR."},{"key":"10.1016\/j.cviu.2026.104791_b198","unstructured":"Wang,\u00a0P., Zheng,\u00a0W., Chen,\u00a0T., Wang,\u00a0Z., 2022b. Anti-oversmoothing in deep vision transformers via the fourier domain analysis: From theory to practice. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b199","doi-asserted-by":"crossref","unstructured":"Wei,\u00a0C., Duke,\u00a0B., Jiang,\u00a0R., Aarabi,\u00a0P., Taylor,\u00a0G., Shkurti,\u00a0F., 2023. Sparsifiner: Learning sparse instance-dependent attention for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 22680\u201322689.","DOI":"10.1109\/CVPR52729.2023.02172"},{"key":"10.1016\/j.cviu.2026.104791_b200","article-title":"Using the nystr\u00f6m method to speed up kernel machines","volume":"13","author":"Williams","year":"2000","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b201","doi-asserted-by":"crossref","first-page":"12760","DOI":"10.1109\/TPAMI.2022.3202765","article-title":"P2t: Pyramid pooling transformer for scene understanding","volume":"45","author":"Wu","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b202","doi-asserted-by":"crossref","unstructured":"Wu,\u00a0S., Wu,\u00a0T., Tan,\u00a0H., Guo,\u00a0G., 2022a. Pale transformer: A general vision transformer backbone with pale-shaped attention. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 2731\u20132739.","DOI":"10.1609\/aaai.v36i3.20176"},{"key":"10.1016\/j.cviu.2026.104791_b203","doi-asserted-by":"crossref","unstructured":"Wu,\u00a0H., Xiao,\u00a0B., Codella,\u00a0N., Liu,\u00a0M., Dai,\u00a0X., Yuan,\u00a0L., Zhang,\u00a0L., 2021. Cvt: Introducing convolutions to vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 22\u201331.","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"10.1016\/j.cviu.2026.104791_b204","doi-asserted-by":"crossref","first-page":"11120","DOI":"10.1109\/TPAMI.2023.3265499","article-title":"Pslt: a light-weight vision transformer with ladder self-attention and progressive shift","volume":"45","author":"Wu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b205","doi-asserted-by":"crossref","first-page":"69925","DOI":"10.52202\/079017-2235","article-title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","volume":"37","author":"Wu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b206","doi-asserted-by":"crossref","unstructured":"Xia,\u00a0Z., Pan,\u00a0X., Song,\u00a0S., Li,\u00a0L., Huang,\u00a0G., 2022. Vision transformer with deformable attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4794\u20134803.","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"10.1016\/j.cviu.2026.104791_b207","unstructured":"Xiao,\u00a0C., Li,\u00a0M., ZHANG,\u00a0Z., Meng,\u00a0D., Zhang,\u00a0L., 2025a. Spatial-mamba: Effective visual state space models via structure-aware state fusion. In: The Thirteenth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b208","doi-asserted-by":"crossref","DOI":"10.1016\/j.iopt.2025.100003","article-title":"Optical image processing and applications empowered by vision-language models","volume":"1","author":"Xiao","year":"2025","journal-title":"IOptics"},{"key":"10.1016\/j.cviu.2026.104791_b209","doi-asserted-by":"crossref","unstructured":"Xiao,\u00a0B., Wu,\u00a0H., Xu,\u00a0W., Dai,\u00a0X., Hu,\u00a0H., Lu,\u00a0Y., Zeng,\u00a0M., Liu,\u00a0C., Yuan,\u00a0L., 2024. Florence-2: Advancing a unified representation for a variety of vision tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4818\u20134829.","DOI":"10.1109\/CVPR52733.2024.00461"},{"key":"10.1016\/j.cviu.2026.104791_b210","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0F., Nie,\u00a0J., Tang,\u00a0Y., Zhang,\u00a0W., Zhao,\u00a0H., 2025. Mamba-adaptor: State space model adaptor for visual recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 20124\u201320134.","DOI":"10.1109\/CVPR52734.2025.01874"},{"key":"10.1016\/j.cviu.2026.104791_b211","unstructured":"Xie,\u00a0E., Wang,\u00a0W., Yu,\u00a0Z., Anandkumar,\u00a0A., Alvarez,\u00a0J., Luo,\u00a0P., 2021. Segformer: Simple and efficient design for semantic segmentation with transformers. In: Advances in Neural Information Processing Systems. pp. 12077\u201312090."},{"key":"10.1016\/j.cviu.2026.104791_b212","doi-asserted-by":"crossref","unstructured":"Xiong,\u00a0T., Liew,\u00a0J., Huang,\u00a0Z., Feng,\u00a0J., Liu,\u00a0X., 2025. Gigatok: Scaling visual tokenizers to 3 billion parameters for autoregressive image generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 18770\u201318780.","DOI":"10.1109\/ICCV51701.2025.01744"},{"key":"10.1016\/j.cviu.2026.104791_b213","doi-asserted-by":"crossref","unstructured":"Xu,\u00a0W., Xu,\u00a0Y., Chang,\u00a0T., Tu,\u00a0Z., 2021a. Co-scale conv-attentional image transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9981\u20139990.","DOI":"10.1109\/ICCV48922.2021.00983"},{"key":"10.1016\/j.cviu.2026.104791_b214","doi-asserted-by":"crossref","unstructured":"Xu,\u00a0Y., Zhang,\u00a0Z., Zhang,\u00a0M., Sheng,\u00a0K., Li,\u00a0K., Dong,\u00a0W., Zhang,\u00a0L., Xu,\u00a0C., Sun,\u00a0X., 2022. Evo-vit: Slow-fast token evolution for dynamic vision transformer. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 2964\u20132972.","DOI":"10.1609\/aaai.v36i3.20202"},{"key":"10.1016\/j.cviu.2026.104791_b215","first-page":"28522","article-title":"Vitae: Vision transformer advanced by exploring intrinsic inductive bias","volume":"34","author":"Xu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b216","doi-asserted-by":"crossref","unstructured":"Xue,\u00a0L., Shu,\u00a0M., Awadalla,\u00a0A., Wang,\u00a0J., Yan,\u00a0A., Purushwalkam,\u00a0S., Zhou,\u00a0H., Prabhu,\u00a0V., Dai,\u00a0Y., Ryoo,\u00a0M., et al., 2025. Blip-3: A family of open large multimodal models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6124\u20136135.","DOI":"10.1109\/ICCVW69036.2025.00644"},{"key":"10.1016\/j.cviu.2026.104791_b217","unstructured":"Yang,\u00a0J., Li,\u00a0C., Zhang,\u00a0P., Dai,\u00a0X., Xiao,\u00a0B., Yuan,\u00a0L., Gao,\u00a0J., 2021. Focal attention for long-range interactions in vision transformers. In: Beygelzimer,\u00a0A., Dauphin,\u00a0Y., Liang,\u00a0P., Vaughan,\u00a0J. (Eds.), Advances in Neural Information Processing Systems."},{"key":"10.1016\/j.cviu.2026.104791_b218","series-title":"European Conference on Computer Vision","first-page":"480","article-title":"Scalablevit: Rethinking the context-oriented generalization of vision transformer","author":"Yang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b219","doi-asserted-by":"crossref","unstructured":"Yang,\u00a0C., Wang,\u00a0Y., Zhang,\u00a0J., Zhang,\u00a0H., Wei,\u00a0Z., Lin,\u00a0Z., Yuille,\u00a0A., 2022a. Lite vision transformer with enhanced self-attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11998\u201312008.","DOI":"10.1109\/CVPR52688.2022.01169"},{"key":"10.1016\/j.cviu.2026.104791_b220","doi-asserted-by":"crossref","first-page":"10870","DOI":"10.1109\/TPAMI.2023.3268446","article-title":"Dual vision transformer","volume":"45","author":"Yao","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b221","series-title":"European Conference on Computer Vision","first-page":"328","article-title":"Wave-vit: Unifying wavelet and transformers for visual representation learning","author":"Yao","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b222","doi-asserted-by":"crossref","unstructured":"Yin,\u00a0H., Vahdat,\u00a0A., Alvarez,\u00a0J., Mallya,\u00a0A., Kautz,\u00a0J., Molchanov,\u00a0P., 2022. A-vit: Adaptive tokens for efficient vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10809\u201310818.","DOI":"10.1109\/CVPR52688.2022.01054"},{"key":"10.1016\/j.cviu.2026.104791_b223","unstructured":"You,\u00a0Y., Li,\u00a0J., Reddi,\u00a0S., Hseu,\u00a0J., Kumar,\u00a0S., Bhojanapalli,\u00a0S., Song,\u00a0X., Demmel,\u00a0J., Keutzer,\u00a0K., Hsieh,\u00a0C., 2019. Large batch optimization for deep learning: Training bert in 76 minutes. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b224","doi-asserted-by":"crossref","unstructured":"You,\u00a0H., Xiong,\u00a0Y., Dai,\u00a0X., Wu,\u00a0B., Zhang,\u00a0P., Fan,\u00a0H., Vajda,\u00a0P., Lin,\u00a0Y., 2023. Castling-vit: Compressing self-attention via switching towards linear-angular attention at vision transformer inference. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 14431\u201314442.","DOI":"10.1109\/CVPR52729.2023.01387"},{"key":"10.1016\/j.cviu.2026.104791_b225","unstructured":"Yu,\u00a0J., Li,\u00a0X., Koh,\u00a0J., Zhang,\u00a0H., Pang,\u00a0R., Qin,\u00a0J., Ku,\u00a0A., Xu,\u00a0Y., Baldridge,\u00a0J., Wu,\u00a0Y., 2022. Vector-quantized image modeling with improved VQGAN. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b226","series-title":"Frequency autoregressive image generation with continuous tokens","author":"Yu","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b227","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"4484","article-title":"Mambaout: Do we really need mamba for vision?","author":"Yu","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b228","first-page":"12992","article-title":"Glance-and-gaze vision transformer","volume":"34","author":"Yu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b229","series-title":"Florence: A new foundation model for computer vision","author":"Yuan","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b230","doi-asserted-by":"crossref","unstructured":"Yuan,\u00a0L., Chen,\u00a0Y., Wang,\u00a0T., Yu,\u00a0W., Shi,\u00a0Y., Jiang,\u00a0Z., Tay,\u00a0F., Feng,\u00a0J., Yan,\u00a0S., 2021c. Tokens-to-token vit: Training vision transformers from scratch on imagenet. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. ICCV, pp. 558\u2013567.","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"10.1016\/j.cviu.2026.104791_b231","unstructured":"Yuan,\u00a0Y., Fu,\u00a0R., Huang,\u00a0L., Lin,\u00a0W., Zhang,\u00a0C., Chen,\u00a0X., Wang,\u00a0J., 2021d. Hrformer: high-resolution transformer for dense prediction. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. pp. 7281\u20137293."},{"key":"10.1016\/j.cviu.2026.104791_b232","doi-asserted-by":"crossref","unstructured":"Yuan,\u00a0K., Guo,\u00a0S., Liu,\u00a0Z., Zhou,\u00a0A., Yu,\u00a0F., Wu,\u00a0W., 2021a. Incorporating convolution designs into visual transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 579\u2013588.","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"10.1016\/j.cviu.2026.104791_b233","first-page":"6575","article-title":"Volo: Vision outlooker for visual recognition","volume":"45","author":"Yuan","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b234","series-title":"Delta-llava: Base-then-specialize alignment for token-efficient vision-language models","author":"Zamini","year":"2025"},{"key":"10.1016\/j.cviu.2026.104791_b235","doi-asserted-by":"crossref","unstructured":"Zamir,\u00a0S., Arora,\u00a0A., Khan,\u00a0S., Hayat,\u00a0M., Khan,\u00a0F., Yang,\u00a0M., 2022. Restormer: Efficient transformer for high-resolution image restoration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 5728\u20135739.","DOI":"10.1109\/CVPR52688.2022.00564"},{"key":"10.1016\/j.cviu.2026.104791_b236","doi-asserted-by":"crossref","unstructured":"Zhai,\u00a0X., Mustafa,\u00a0B., Kolesnikov,\u00a0A., Beyer,\u00a0L., 2023. Sigmoid loss for language image pre-training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 11975\u201311986.","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"10.1016\/j.cviu.2026.104791_b237","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing","first-page":"375","article-title":"On orthogonality constraints for transformers","author":"Zhang","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b238","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0P., Dai,\u00a0X., Yang,\u00a0J., Xiao,\u00a0B., Yuan,\u00a0L., Zhang,\u00a0L., Gao,\u00a0J., 2021b. Multi-scale vision longformer: A new vision transformer for high-resolution image encoding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2998\u20133008.","DOI":"10.1109\/ICCV48922.2021.00299"},{"key":"10.1016\/j.cviu.2026.104791_b239","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"11304","article-title":"Styleswin: Transformer-based gan for high-resolution image generation","author":"Zhang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b240","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"1389","article-title":"Rethinking mobile block for efficient attention-based models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104791_b241","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0J., Nguyen,\u00a0A., Han,\u00a0X., Trinh,\u00a0V., Qin,\u00a0H., Samaras,\u00a0D., Hosseini,\u00a0M., 2025. 2dmamba: Efficient state space model for image representation with applications on giga-pixel whole slide image classification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 3583\u20133592.","DOI":"10.1109\/CVPR52734.2025.00339"},{"key":"10.1016\/j.cviu.2026.104791_b242","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0J., Peng,\u00a0H., Wu,\u00a0K., Liu,\u00a0M., Xiao,\u00a0B., Fu,\u00a0J., Yuan,\u00a0L., 2022c. Minivit: Compressing vision transformers with weight multiplexing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 12145\u201312154.","DOI":"10.1109\/CVPR52688.2022.01183"},{"key":"10.1016\/j.cviu.2026.104791_b243","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0X., Tan,\u00a0R., 2025. Mamba as a bridge: Where vision foundation models meet vision language models for domain-generalized semantic segmentation. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 14527\u201314537.","DOI":"10.1109\/CVPR52734.2025.01354"},{"key":"10.1016\/j.cviu.2026.104791_b244","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0B., Tian,\u00a0Z., Tang,\u00a0Q., Chu,\u00a0X., Wei,\u00a0X., Shen,\u00a0C., et al., 2022b. Segvit: Semantic segmentation with plain vision transformers. In: Advances in Neural Information Processing Systems. pp. 4971\u20134982.","DOI":"10.52202\/068431-0359"},{"key":"10.1016\/j.cviu.2026.104791_b245","first-page":"1","article-title":"Tranmamba: a lightweight hybrid transformer-mamba network for single image super-resolution","volume":"19","author":"Zhang","year":"2025","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.cviu.2026.104791_b246","series-title":"European Conference on Computer Vision","first-page":"466","article-title":"Vsa: Learning varied-size window attention in vision transformers","author":"Zhang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b247","unstructured":"Zhang,\u00a0Q., Yang,\u00a0Y., 2021. Rest: An efficient transformer for visual recognition. In: Advances in Neural Information Processing Systems. pp. 15475\u201315485."},{"key":"10.1016\/j.cviu.2026.104791_b248","doi-asserted-by":"crossref","first-page":"3608","DOI":"10.1109\/TPAMI.2023.3347693","article-title":"Vision transformer with quadrangle attention","volume":"46","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b249","series-title":"European Conference on Computer Vision","first-page":"350","article-title":"Detecting twenty-thousand classes using image-level supervision","author":"Zhou","year":"2022"},{"key":"10.1016\/j.cviu.2026.104791_b250","first-page":"12738","article-title":"Token selection is a simple booster for vision transformers","volume":"45","author":"Zhou","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104791_b251","series-title":"Deepvit: Towards deeper vision transformer","author":"Zhou","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b252","series-title":"Refiner: Refining self-attention for vision transformers","author":"Zhou","year":"2021"},{"key":"10.1016\/j.cviu.2026.104791_b253","doi-asserted-by":"crossref","first-page":"2516","DOI":"10.1007\/s11263-023-01813-x","article-title":"What limits the performance of local self-attention?","volume":"131","author":"Zhou","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.cviu.2026.104791_b254","unstructured":"Zhou,\u00a0C., YU,\u00a0L., Babu,\u00a0A., Tirumala,\u00a0K., Yasunaga,\u00a0M., Shamis,\u00a0L., Kahn,\u00a0J., Ma,\u00a0X., Zettlemoyer,\u00a0L., Levy,\u00a0O., 2025a. Transfusion: Predict the next token and diffuse images with one multi-modal model. In: The Thirteenth International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104791_b255","doi-asserted-by":"crossref","unstructured":"Zhou,\u00a0S., Zeng,\u00a0H., Lu,\u00a0Y., Shao,\u00a0T., Tang,\u00a0K., Chen,\u00a0Y., Liu,\u00a0J., Su,\u00a0J., 2025b. Binarized mamba-transformer for lightweight quad bayer hybridevs demosaicing. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 8817\u20138827.","DOI":"10.1109\/CVPR52734.2025.00824"},{"key":"10.1016\/j.cviu.2026.104791_b256","doi-asserted-by":"crossref","unstructured":"Zhou,\u00a0H., Zhang,\u00a0S., Peng,\u00a0J., Zhang,\u00a0S., Li,\u00a0J., Xiong,\u00a0H., Zhang,\u00a0W., 2021c. Informer: Beyond efficient transformer for long sequence time-series forecasting. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 11106\u201311115.","DOI":"10.1609\/aaai.v35i12.17325"},{"key":"10.1016\/j.cviu.2026.104791_b257","unstructured":"Zhu,\u00a0L., Liao,\u00a0B., Zhang,\u00a0Q., Wang,\u00a0X., Liu,\u00a0W., Wang,\u00a0X., 2024a. Vision mamba: Efficient visual representation learning with bidirectional state space model. In: Forty-First International Conference on Machine Learning."},{"key":"10.1016\/j.cviu.2026.104791_b258","doi-asserted-by":"crossref","unstructured":"Zhu,\u00a0R., Pan,\u00a0Y., Li,\u00a0Y., Yao,\u00a0T., Sun,\u00a0Z., Mei,\u00a0T., Chen,\u00a0C., 2024c. Sd-dit: Unleashing the power of self-supervised discrimination in diffusion transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8435\u20138445.","DOI":"10.1109\/CVPR52733.2024.00806"},{"key":"10.1016\/j.cviu.2026.104791_b259","first-page":"17723","article-title":"Long-short transformer: Efficient transformers for language and vision (lst)","volume":"34","author":"Zhu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104791_b260","doi-asserted-by":"crossref","unstructured":"Zhu,\u00a0L., Wang,\u00a0X., Ke,\u00a0Z., Zhang,\u00a0W., Lau,\u00a0R., 2023. Biformer: Vision transformer with bi-level routing attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10323\u201310333.","DOI":"10.1109\/CVPR52729.2023.00995"},{"key":"10.1016\/j.cviu.2026.104791_b261","doi-asserted-by":"crossref","first-page":"42941","DOI":"10.52202\/079017-1359","article-title":"Revisiting the integration of convolution and attention for vision backbone","volume":"37","author":"Zhu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S107731422600158X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S107731422600158X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T06:09:51Z","timestamp":1783058991000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S107731422600158X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":261,"alternative-id":["S107731422600158X"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104791","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Self-attention as the backbone: A survey on Vision Transformers","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104791","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104791"}}