{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T16:57:23Z","timestamp":1780765043862,"version":"3.54.1"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T00:00:00Z","timestamp":1748217600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T00:00:00Z","timestamp":1748217600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001665","name":"Agence Nationale de la Recherche","doi-asserted-by":"publisher","award":["ANR-18-CE23-0011"],"award-info":[{"award-number":["ANR-18-CE23-0011"]}],"id":[{"id":"10.13039\/501100001665","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s11263-025-02475-7","type":"journal-article","created":{"date-parts":[[2025,5,26]],"date-time":"2025-05-26T07:40:32Z","timestamp":1748245232000},"page":"6129-6144","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Lightweight Structure-Aware Attention for Visual Understanding"],"prefix":"10.1007","volume":"133","author":[{"given":"Heeseung","family":"Kwon","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Francisco M.","family":"Castro","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9294-6714","authenticated-orcid":false,"given":"Manuel J.","family":"Marin-Jimenez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nicolas","family":"Guil","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1838-5936","authenticated-orcid":false,"given":"Karteek","family":"Alahari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,26]]},"reference":[{"key":"2475_CR1","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. & et al (2020). An image is worth 16x16 words: Transformers for image recognition at scale. Proc. International Conference on Learning Representations (ICLR)"},{"key":"2475_CR2","unstructured":"Li, K., Wang, Y., Gao, P., Song, G., Liu, Y., Li, H. & Qiao, Y. (2022). Uniformer: Unified transformer for efficient spatiotemporal representation learning. Proc. International Conference on Learning Representations (ICLR)"},{"key":"2475_CR3","doi-asserted-by":"crossref","unstructured":"Yuan, L., Chen, Y., Wang, T., Yu, W., Shi, Y., Jiang, Z.-H., Tay, F.E., Feng, J. & Yan, S. (2021). Tokens-to-token vit: Training vision transformers from scratch on imagenet. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 558\u2013567","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"2475_CR4","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A. & J\u00e9gou, H. (2021). Training data-efficient image transformers & distillation through attention. In: Proc. International Conference on Machine Learning (ICML), pp. 10347\u201310357. PMLR"},{"key":"2475_CR5","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W. & Liu, P.J. (2019). Exploring the limits of transfer learning with a unified text-to-text transformer. arXiv preprint arXiv:1910.10683"},{"key":"2475_CR6","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S. & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2475_CR7","unstructured":"Bello, I. (2020). Lambdanetworks: Modeling long-range interactions without attention. In: Proc. International Conference on Learning Representations (ICLR)"},{"key":"2475_CR8","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jia, J. & Koltun, V. (2020). Exploring self-attention for image recognition. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10076\u201310085","DOI":"10.1109\/CVPR42600.2020.01009"},{"key":"2475_CR9","first-page":"8046","volume":"34","author":"M Kim","year":"2021","unstructured":"Kim, M., Kwon, H., Wang, C., Kwak, S., & Cho, M. (2021). Relational self-attention: What\u2019s missing in attention for video understanding. Proc. Neural Information Processing Systems (NeurIPS), 34, 8046\u20138059.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR10","unstructured":"Wang, S., Li, B., Khabsa, M., Fang, H. & Ma, H. (2020). Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768"},{"key":"2475_CR11","unstructured":"Choromanski, K., Likhosherstov, V., Dohan, D., Song, X., Gane, A., Sarlos, T., Hawkins, P., Davis, J., Mohiuddin, A., Kaiser, L. & et al. (2021). Rethinking attention with performers. In: Proc. International Conference on Learning Representations (ICLR)"},{"key":"2475_CR12","unstructured":"Qin, Z., Sun, W., Deng, H., Li, D., Wei, Y., Lv, B., Yan, J., Kong, L. & Zhong, Y. (2022). cosformer: Rethinking softmax in attention. arXiv preprint arXiv:2202.08791"},{"key":"2475_CR13","unstructured":"Liutkus, A., C\u0131fka, O., Wu, S.-L., Simsekli, U., Yang, Y.-H. & Richard, G. (2021). Relative positional encoding for transformers with linear complexity. In: Proc. International Conference on Machine Learning (ICML), pp. 7067\u20137079. PMLR"},{"key":"2475_CR14","doi-asserted-by":"crossref","unstructured":"Chen, P. (2021). Permuteformer: Efficient relative position encoding for long sequences. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 10606\u201310618","DOI":"10.18653\/v1\/2021.emnlp-main.828"},{"key":"2475_CR15","first-page":"22795","volume":"34","author":"S Luo","year":"2021","unstructured":"Luo, S., Li, S., Cai, T., He, D., Peng, D., Zheng, S., Ke, G., Wang, L., & Liu, T.-Y. (2021). Stable, fast and accurate: Kernelized attention with relative positional encoding. Proc. Neural Information Processing Systems (NeurIPS), 34, 22795\u201322807.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR16","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C.-Y., Fan, H., Mangalam, K., Xiong, B., Malik, J. & Feichtenhofer, C. (2022). Mvitv2: Improved multiscale vision transformers for classification and detection. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4804\u20134814","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"2475_CR17","doi-asserted-by":"crossref","unstructured":"Wu, H., Xiao, B., Codella, N., Liu, M., Dai, X., Yuan, L. & Zhang, L. (2021). Cvt: Introducing convolutions to vision transformers. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 22\u201331","DOI":"10.1109\/ICCV48922.2021.00009"},{"issue":"3","key":"2475_CR18","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","volume":"8","author":"W Wang","year":"2022","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P., & Shao, L. (2022). Pvt v2: Improved baselines with pyramid vision transformer. Computational Visual Media, 8(3), 415\u2013424.","journal-title":"Computational Visual Media"},{"key":"2475_CR19","first-page":"23495","volume":"35","author":"C Si","year":"2022","unstructured":"Si, C., Yu, W., Zhou, P., Zhou, Y., Wang, X., & Yan, S. (2022). Inception transformer. Proc. Neural Information Processing Systems (NeurIPS), 35, 23495\u201323509.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR20","doi-asserted-by":"crossref","unstructured":"Tu, Z., Talebi, H., Zhang, H., Yang, F., Milanfar, P., Bovik, A. & Li, Y. (2022). Maxvit: Multi-axis vision transformer. In: Proc. European Conference on Computer Vision (ECCV), pp. 459\u2013479. Springer","DOI":"10.1007\/978-3-031-20053-3_27"},{"key":"2475_CR21","doi-asserted-by":"crossref","unstructured":"Shechtman, E. & Irani, M. (2007). Matching local self-similarities across images and videos. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1\u20138. IEEE","DOI":"10.1109\/CVPR.2007.383198"},{"key":"2475_CR22","doi-asserted-by":"crossref","unstructured":"Wang, H., Tran, D., Torresani, L. & Feiszli, M. (2020). Video modeling with correlation networks. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 352\u2013361","DOI":"10.1109\/CVPR42600.2020.00043"},{"key":"2475_CR23","doi-asserted-by":"crossref","unstructured":"Kwon, H., Kim, M., Kwak, S. & Cho, M. (2021). Learning self-similarity in space and time as generalized motion for video action recognition. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 13065\u201313075","DOI":"10.1109\/ICCV48922.2021.01282"},{"key":"2475_CR24","doi-asserted-by":"crossref","unstructured":"Sun, D., Yang, X., Liu, M.-Y. & Kautz, J. (2018). Pwc-net: Cnns for optical flow using pyramid, warping, and cost volume. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8934\u20138943","DOI":"10.1109\/CVPR.2018.00931"},{"key":"2475_CR25","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K. & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 248\u2013255. IEEE","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2475_CR26","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P. & et al. (2017). The kinetics human action video dataset. arXiv preprint arXiv:1705.06950"},{"key":"2475_CR27","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P. & Zitnick, C.L. (2014). Microsoft coco: Common objects in context. In: Proc. European Conference on Computer Vision (ECCV), pp. 740\u2013755. Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2475_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A. & Torralba, A. (2017). Scene parsing through ade20k dataset. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 633\u2013641","DOI":"10.1109\/CVPR.2017.544"},{"key":"2475_CR29","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A. & Zagoruyko, S. (2020). End-to-end object detection with transformers. In: Proc. European Conference on Computer Vision (ECCV), pp. 213\u2013229. Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2475_CR30","doi-asserted-by":"crossref","unstructured":"Strudel, R., Garcia, R., Laptev, I. & Schmid, C.(2021). Segmenter: Transformer for semantic segmentation. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 7262\u20137272","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"2475_CR31","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M. & Schmid, C. (2021). Vivit: A video vision transformer. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"2475_CR32","doi-asserted-by":"crossref","unstructured":"Vaswani, A., Ramachandran, P., Srinivas, A., Parmar, N., Hechtman, B. & Shlens, J. (2021). Scaling local self-attention for parameter efficient visual backbones. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12894\u201312904","DOI":"10.1109\/CVPR46437.2021.01270"},{"key":"2475_CR33","first-page":"30392","volume":"34","author":"T Xiao","year":"2021","unstructured":"Xiao, T., Singh, M., Mintun, E., Darrell, T., Doll\u00e1r, P., & Girshick, R. (2021). Early convolutions help transformers see better. Proc. Neural Information Processing Systems (NeurIPS), 34, 30392\u201330400.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR34","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han, K., Xiao, A., Wu, E., Guo, J., Xu, C., & Wang, Y. (2021). Transformer in transformer. Proc. Neural Information Processing Systems (NeurIPS), 34, 15908\u201315919.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, H., Zhao, L., Chen, T., Arik, S.\u00d6. & Pfister, T. (2022). Nested hierarchical transformer: Towards accurate, data-efficient and interpretable visual understanding. In: Proc. AAAI Conference on Artificial Intelligence (AAAI), vol. 36, pp. 3417\u20133425","DOI":"10.1609\/aaai.v36i3.20252"},{"key":"2475_CR36","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Chen, D., Zhang, W., Yu, N., Yuan, L., Chen, D. & Guo, B. (2022). Cswin transformer: A general vision transformer backbone with cross-shaped windows. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12124\u201312134","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"2475_CR37","first-page":"3965","volume":"34","author":"Z Dai","year":"2021","unstructured":"Dai, Z., Liu, H., Le, Q. V., & Tan, M. (2021). Coatnet: Marrying convolution and attention for all data sizes. Proc. Neural Information Processing Systems (NeurIPS), 34, 3965\u20133977.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR38","unstructured":"Jia, X., De\u00a0Brabandere, B., Tuytelaars, T. & Gool, L.V. (2016). Dynamic filter networks. Proc. Neural Information Processing Systems (NeurIPS) 29"},{"key":"2475_CR39","doi-asserted-by":"crossref","unstructured":"Li, D., Hu, J., Wang, C., Li, X., She, Q., Zhu, L., Zhang, T. & Chen, Q. (2021). Involution: Inverting the inherence of convolution for visual recognition. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12321\u201312330","DOI":"10.1109\/CVPR46437.2021.01214"},{"key":"2475_CR40","doi-asserted-by":"crossref","unstructured":"Chen, Y., Dai, X., Liu, M., Chen, D., Yuan, L. & Liu, Z. (2020). Dynamic convolution: Attention over convolution kernels. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11030\u201311039","DOI":"10.1109\/CVPR42600.2020.01104"},{"key":"2475_CR41","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Huang, J. & Sun, J. (2020). Weightnet: Revisiting the design space of weight networks. In: Proc. European Conference on Computer Vision (ECCV), pp. 776\u2013792. Springer","DOI":"10.1007\/978-3-030-58555-6_46"},{"key":"2475_CR42","unstructured":"Shen, Z., Zhang, M., Zhao, H., Yi, S. & Li, H. (2021). Efficient attention: Attention with linear complexities. In: Proc. Winter Conference on Applications of Computer Vision (WACV), pp. 3531\u20133539"},{"key":"2475_CR43","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N. & Fleuret, F. (2020). Transformers are rnns: Fast autoregressive transformers with linear attention. In: Proc. International Conference on Machine Learning (ICML), pp. 5156\u20135165. PMLR"},{"key":"2475_CR44","first-page":"980","volume":"34","author":"Y Rao","year":"2021","unstructured":"Rao, Y., Zhao, W., Zhu, Z., Lu, J., & Zhou, J. (2021). Global filter networks for image classification. Proc. Neural Information Processing Systems (NeurIPS), 34, 980\u2013993.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR45","doi-asserted-by":"crossref","unstructured":"Lee-Thorp, J., Ainslie, J., Eckstein, I. & Ontanon, S. (2021). Fnet: Mixing tokens with fourier transforms. arXiv preprint arXiv:2105.03824","DOI":"10.18653\/v1\/2022.naacl-main.319"},{"key":"2475_CR46","doi-asserted-by":"crossref","unstructured":"d\u2019Ascoli, S., Touvron, H., Leavitt, M.L., Morcos, A.S., Biroli, G. & Sagun, L. (2021). Convit: Improving vision transformers with soft convolutional inductive biases. In: Proc. International Conference on Machine Learning (ICML), pp. 2286\u20132296. PMLR","DOI":"10.1088\/1742-5468\/ac9830"},{"key":"2475_CR47","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141. & Polosukhin, I. (2017). Attention is all you need. Proc. Neural Information Processing Systems (NeurIPS) 30"},{"key":"2475_CR48","unstructured":"Ramachandran, P., Parmar, N., Vaswani, A., Bello, I., Levskaya, A. & Shlens, J. (2019). Stand-alone self-attention in vision models. Proc. Neural Information Processing Systems (NeurIPS) 32"},{"issue":"2","key":"2475_CR49","first-page":"171","volume":"74","author":"G Strang","year":"1986","unstructured":"Strang, G. (1986). A proposal for toeplitz matrix calculations. Applied Mathematics, 74(2), 171\u2013176.","journal-title":"Applied Mathematics"},{"key":"2475_CR50","unstructured":"Fu, D.Y., Dao, T., Saab, K.K., Thomas, A.W., Rudra, A. & Re, C. (2022). Hungry hungry hippos: Towards language modeling with state space models. In: Proc. International Conference on Learning Representations (ICLR)"},{"issue":"3","key":"2475_CR51","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., Huang, Z., Karpathy, A., Khosla, A., Bernstein, M., Berg, A. C., & Fei-Fei, L. (2015). ImageNet Large Scale Visual Recognition Challenge. International Journal of Computer Vision (IJCV), 115(3), 211\u2013252. https:\/\/doi.org\/10.1007\/s11263-015-0816-y","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"2475_CR52","doi-asserted-by":"crossref","unstructured":"Lin, W., Wu, Z., Chen, J., Huang, J. & Jin, L. (2023). Scale-aware modulation meet transformer. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 6015\u20136026","DOI":"10.1109\/ICCV51070.2023.00553"},{"issue":"10","key":"2475_CR53","doi-asserted-by":"publisher","first-page":"12581","DOI":"10.1109\/TPAMI.2023.3282631","volume":"45","author":"K Li","year":"2023","unstructured":"Li, K., Wang, Y., Zhang, J., Gao, P., Song, G., Liu, Y., Li, H., & Qiao, Y. (2023). Uniformer: Unifying convolution and self-attention for visual recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 45(10), 12581\u201312600.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"2475_CR54","doi-asserted-by":"crossref","unstructured":"Wang, W., Dai, J., Chen, Z., Huang, Z., Li, Z., Zhu, X., Hu, X., Lu, T., Lu, L., Li, H. & et al. (2023). Internimage: Exploring large-scale vision foundation models with deformable convolutions. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 14408\u201314419","DOI":"10.1109\/CVPR52729.2023.01385"},{"key":"2475_CR55","doi-asserted-by":"crossref","unstructured":"Pan, X., Ye, T., Xia, Z., Song, S. & Huang, G. (2023). Slide-transformer: Hierarchical vision transformer with local self-attention. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2082\u20132091","DOI":"10.1109\/CVPR52729.2023.00207"},{"key":"2475_CR56","unstructured":"Chu, X., Tian, Z., Zhang, B., Wang, X. & Shen, C. (2023). Conditional positional encodings for vision transformers. In: Proc. International Conference on Learning Representations (ICLR)"},{"key":"2475_CR57","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M. & Adam, H. (2017). Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861"},{"key":"2475_CR58","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S. & Sun, J. (2016). Deep residual learning for image recognition. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"2475_CR59","doi-asserted-by":"crossref","unstructured":"Radosavovic, I., Kosaraju, R.P., Girshick, R., He, K. & Doll\u00e1r, P. (2020). Designing network design spaces. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10428\u201310436","DOI":"10.1109\/CVPR42600.2020.01044"},{"key":"2475_CR60","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T. & Xie, S. (2022). A convnet for the 2020s. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11976\u201311986","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2475_CR61","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P. & Shao, L. (2021). Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 568\u2013578","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"2475_CR62","unstructured":"Guo, M.-H., Lu, C.-Z., Liu, Z.-N., Cheng, M.-M. & Hu, S.-M. (2022). Visual attention network. arXiv preprint arXiv:2202.09741"},{"key":"2475_CR63","unstructured":"Vasu, P.K.A., Gabriel, J., Zhu, J., Tuzel, O. & Ranjan, A. (2023). Fastvit: A fast hybrid vision transformer using structural reparameterization. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 5785\u20135795"},{"key":"2475_CR64","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C. (2020). X3d: Expanding architectures for efficient video recognition. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"2475_CR65","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J. & He, K. (2019). Slowfast networks for video recognition. In: Proc. IEEE International Conference on Computer Vision (ICCV)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"2475_CR66","first-page":"19594","volume":"34","author":"A Bulat","year":"2021","unstructured":"Bulat, A., Perez Rua, J. M., Sudhakaran, S., Martinez, B., & Tzimiropoulos, G. (2021). Space-time mixing attention for video transformer. Proc. Neural Information Processing Systems (NeurIPS), 34, 19594\u201319607.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR67","first-page":"12493","volume":"34","author":"M Patrick","year":"2021","unstructured":"Patrick, M., Campbell, D., Asano, Y., Misra, I., Metze, F., Feichtenhofer, C., Vedaldi, A., & Henriques, J. F. (2021). Keeping your eye on the ball: Trajectory attention in video transformers. Proc. Neural Information Processing Systems (NeurIPS), 34, 12493\u201312506.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR68","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S. & Hu, H. (2022). Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"2475_CR69","unstructured":"Bertasius, G., Wang, H. & Torresani, L. (2021). Is space-time attention all you need for video understanding? In: Proc. International Conference on Machine Learning (ICML), vol. 2, p. 4"},{"key":"2475_CR70","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J. & Feichtenhofer, C. (2021). Multiscale vision transformers. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 6824\u20136835","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"2475_CR71","first-page":"9355","volume":"34","author":"X Chu","year":"2021","unstructured":"Chu, X., Tian, Z., Wang, Y., Zhang, B., Ren, H., Wei, X., Xia, H., & Shen, C. (2021). Twins: Revisiting the design of spatial attention in vision transformers. Proc. Neural Information Processing Systems (NeurIPS), 34, 9355\u20139366.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR72","doi-asserted-by":"crossref","unstructured":"Zhang, P., Dai, X., Yang, J., Xiao, B., Yuan, L., Zhang, L. & Gao, J. (2021). Multi-scale vision longformer: A new vision transformer for high-resolution image encoding. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 2998\u20133008","DOI":"10.1109\/ICCV48922.2021.00299"},{"key":"2475_CR73","first-page":"30008","volume":"34","author":"J Yang","year":"2021","unstructured":"Yang, J., Li, C., Zhang, P., Dai, X., Xiao, B., Yuan, L., & Gao, J. (2021). Focal attention for long-range interactions in vision transformers. Proc. Neural Information Processing Systems (NeurIPS), 34, 30008\u201330022.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR74","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P. & Girshick, R. (2017). Mask r-cnn. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2475_CR75","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Girshick, R., He, K. & Doll\u00e1r, P. (2019). Panoptic feature pyramid networks. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6399\u20136408.","DOI":"10.1109\/CVPR.2019.00656"},{"key":"2475_CR76","unstructured":"Loshchilov, I. & Hutter, F. (2018). Decoupled weight decay regularization. In: Proc. International Conference on Learning Representations (ICLR)."},{"key":"2475_CR77","unstructured":"Zhang, H., Cisse, M., Dauphin, Y.N. & Lopez-Paz, D. (2018). mixup: Beyond empirical risk minimization. In: Proc. International Conference on Learning Representations (ICLR)."},{"key":"2475_CR78","doi-asserted-by":"crossref","unstructured":"Yun, S., Han, D., Oh, S.J., Chun, S., Choe, J. & Yoo, Y. (2019). Cutmix: Regularization strategy to train strong classifiers with localizable features. In: Proc. IEEE International Conference on Computer Vision (ICCV), pp. 6023\u20136032.","DOI":"10.1109\/ICCV.2019.00612"},{"key":"2475_CR79","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J. & Wojna, Z. (2016). Rethinking the inception architecture for computer vision. In: Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2818\u20132826.","DOI":"10.1109\/CVPR.2016.308"},{"key":"2475_CR80","doi-asserted-by":"crossref","unstructured":"Huang, G., Sun, Y., Liu, Z., Sedra, D. & Weinberger, K.Q. (2016). Deep networks with stochastic depth. In: Proc. European Conference on Computer Vision (ECCV), pp. 646\u2013661. Springer","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"2475_CR81","first-page":"18613","volume":"33","author":"ED Cubuk","year":"2020","unstructured":"Cubuk, E. D., Zoph, B., Shlens, J., & Le, Q. (2020). Randaugment: Practical automated data augmentation with a reduced search space. Proc. Neural Information Processing Systems (NeurIPS), 33, 18613\u201318624.","journal-title":"Proc. Neural Information Processing Systems (NeurIPS)"},{"key":"2475_CR82","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Zheng, L., Kang, G., Li, S. & Yang, Y. (2020). Random erasing data augmentation. In: Proc. AAAI Conference on Artificial Intelligence (AAAI), vol. 34, pp. 13001\u201313008.","DOI":"10.1609\/aaai.v34i07.7000"},{"key":"2475_CR83","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A. & He, K. (2018). Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803.","DOI":"10.1109\/CVPR.2018.00813"},{"key":"2475_CR84","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X. & Van\u00a0Gool, L. (2016). Temporal segment networks: Towards good practices for deep action recognition. In: Proc. European Conference on Computer Vision (ECCV), pp. 20\u201336. Springer","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"2475_CR85","unstructured":"Fu, D.Y., Kumbong, H., Nguyen, E. & R\u00e9, C. (2023). Flashfftconv: Efficient convolutions for long sequences with tensor cores. arXiv preprint arXiv:2311.05908"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02475-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02475-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02475-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T08:02:53Z","timestamp":1757404973000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02475-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,26]]},"references-count":85,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["2475"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02475-7","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,26]]},"assertion":[{"value":"20 October 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}