{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:02:32Z","timestamp":1784894552645,"version":"3.55.0"},"reference-count":75,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100012542","name":"Sichuan Provincial Science and Technology Support Program","doi-asserted-by":"publisher","award":["2025NSFSC2017"],"award-info":[{"award-number":["2025NSFSC2017"]}],"id":[{"id":"10.13039\/100012542","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576295"],"award-info":[{"award-number":["62576295"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Scientic Research Funds project of Science and Technology Department of Sichuan Province","award":["2023YFG0354"],"award-info":[{"award-number":["2023YFG0354"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00530-026-02465-w","type":"journal-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T14:32:52Z","timestamp":1783780372000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["BiCTM: lightweight video recognition via convolutional tube masking and bidirectional motion features"],"prefix":"10.1007","volume":"32","author":[{"given":"Linxi","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhengyan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingwei","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanxi","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingchi","family":"Gui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,11]]},"reference":[{"key":"2465_CR1","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"2465_CR2","doi-asserted-by":"publisher","unstructured":"Wang, L., Tong, Z., Ji, B., Wu, G.: Tdn: Temporal difference networks for efficient action recognition. In. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) vol. 2021, pp. 1895\u20131904 (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.00193","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"2465_CR3","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., Feichtenhofer, C.: Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 6824\u20136835 (2021)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"2465_CR4","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., Paluri, M.: A closer look at spatiotemporal convolutions for action recognition. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp. 6450\u20136459 (2018)","DOI":"10.1109\/CVPR.2018.00675"},{"key":"2465_CR5","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3d: expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"2465_CR6","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., Hu, H.: Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3202\u20133211 (2022)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"2465_CR7","doi-asserted-by":"crossref","unstructured":"Goyal, R., Ebrahimi\u00a0Kahou, S., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., et\u00a0al.: The something something video database for learning and evaluating visual common sense. In: Proceedings of the IEEE international conference on computer vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"2465_CR8","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P.: The kinetics human action video dataset. arXiv preprint (2017). arXiv:1705.06950"},{"key":"2465_CR9","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, p. 25. (2012)"},{"key":"2465_CR10","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., fan, h., Li, Y., He, K.: Masked autoencoders as spatiotemporal learners. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A.: (Eds.), Advances in Neural Information Processing Systems, volume\u00a035, Curran Associates, Inc., pp. 35946\u201335958 (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/e97d1081481a4017df96b51be31001d3-Paper-Conference.pdf","DOI":"10.52202\/068431-2605"},{"key":"2465_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"2465_CR12","doi-asserted-by":"publisher","first-page":"10078","DOI":"10.52202\/068431-0732","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., Wang, L.: Videomae: masked autoencoders are data-efficient learners for self-supervised video pre-training. Adv. Neural. Inf. Process. Syst. 35, 10078\u201310093 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2465_CR13","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Van\u00a0Gool, L.: Temporal segment networks: Towards good practices for deep action recognition. In: European conference on computer vision, Springer, pp. 20\u201336 (2016)","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"2465_CR14","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1007\/978-3-030-58571-6_6","volume-title":"Computer Vision - ECCV 2020","author":"Y Meng","year":"2020","unstructured":"Meng, Y., Lin, C.-C., Panda, R., Sattigeri, P., Karlinsky, L., Oliva, A., Saenko, K., Feris, R.: Ar-net: Adaptive frame resolution for efficient action recognition. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision - ECCV 2020, pp. 86\u2013104. Springer International Publishing, Cham (2020)"},{"key":"2465_CR15","doi-asserted-by":"crossref","unstructured":"Li, H., Wu, Z., Shrivastava, A., Davis, L.S.: 2d or not 2d? adaptive 3d convolution selection for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6155\u20136164 (2021)","DOI":"10.1109\/CVPR46437.2021.00609"},{"key":"2465_CR16","unstructured":"Wu, Z., Xiong, C., Jiang, Y.-G., Davis, L.S.: Liteeval: a coarse-to-fine framework for resource efficient video recognition. In: Wallach, H., Larochelle, H., Beygelzimer, A., d\u2019Alch\u00e9-Buc, F., Fox, E., Garnett, R.: (Eds.), Advances in Neural Information Processing Systems, volume\u00a032, Curran Associates, Inc. (2019a). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2019\/file\/bd853b475d59821e100d3d24303d7747-Paper.pdf"},{"key":"2465_CR17","doi-asserted-by":"crossref","unstructured":"Wu, Z., Xiong, C., Ma, C.-Y., Socher, R., Davis, L.S.: Adaframe: adaptive frame selection for fast video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019b)","DOI":"10.1109\/CVPR.2019.00137"},{"key":"2465_CR18","doi-asserted-by":"crossref","unstructured":"Huang, B., Zhao, Z., Zhang, G., Qiao, Y., Wang, L.: Mgmae: motion guided masking for video masked autoencoding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13493\u201313504 (2023)","DOI":"10.1109\/ICCV51070.2023.01241"},{"key":"2465_CR19","doi-asserted-by":"crossref","unstructured":"Bandara, W.G.C., Patel, N., Gholami, A., Nikkhah, M., Agrawal, M., Patel, V.M.: Adamae: adaptive masking for efficient spatiotemporal learning with masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 14507\u201314517 (2023)","DOI":"10.1109\/CVPR52729.2023.01394"},{"key":"2465_CR20","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-Nouby, A., Singh, M., Alwala, K.V., Joulin, A., Misra, I.: Omnimae: single model masked pretraining on images and videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10406\u201310417 (2023)","DOI":"10.1109\/CVPR52729.2023.01003"},{"key":"2465_CR21","doi-asserted-by":"crossref","unstructured":"Wang, L., Huang, B., Zhao, Z., Tong, Z., He, Y., Wang, Y., Wang, Y., Qiao, Y.: Videomae v2: scaling video masked autoencoders with dual masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR, pp. 14549\u201314560, (2023)","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"2465_CR22","doi-asserted-by":"crossref","unstructured":"Fan, D., Wang, J., Liao, S., Zhu, Y., Bhat, V., Santos-Villalobos, H., MV, R., Li, X.: Motion-guided masking for spatiotemporal representation learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5619\u20135629 (2023)","DOI":"10.1109\/ICCV51070.2023.00517"},{"key":"2465_CR23","doi-asserted-by":"crossref","unstructured":"Sun, X., Chen, P., Chen, L., Li, C., Li, T.H., Tan, M., Gan, C.: Masked motion encoding for self-supervised video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2235\u20132245 (2023)","DOI":"10.1109\/CVPR52729.2023.00222"},{"key":"2465_CR24","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: transformers for image recognition at scale, (2021). arXiv:2010.11929"},{"key":"2465_CR25","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE international conference on computer vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"2465_CR26","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"2465_CR27","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Feiszli, M.: Video classification with channel-separated convolutional networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00565"},{"key":"2465_CR28","doi-asserted-by":"crossref","unstructured":"Xie, S., Sun, C., Huang, J., Tu, Z., Murphy, K.: Rethinking spatiotemporal feature learning: speed-accuracy trade-offs in video classification. In: Proceedings of the European conference on computer vision (ECCV), pp. 305\u2013321 (2018)","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"2465_CR29","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, S., Huang, T.: Multi-scale 3d convolution network for video based person re-identification. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033,, pp. 8618\u20138625 (2019)","DOI":"10.1609\/aaai.v33i01.33018618"},{"key":"2465_CR30","doi-asserted-by":"crossref","unstructured":"Zolfaghari, M., Singh, K., Brox, T.: Eco: efficient convolutional network for online video understanding. In: Proceedings of the European Conference on Computer Vision (ECCV) (2018)","DOI":"10.1007\/978-3-030-01216-8_43"},{"key":"2465_CR31","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In: Advances in Neural Information Processing Systems, p. 27. (2014)"},{"key":"2465_CR32","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Wildes, R.P.: Spatiotemporal multiplier networks for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.787"},{"key":"2465_CR33","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xiong, Y., Lin, D.: Recognize actions by disentangling components of dynamics. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00687"},{"key":"2465_CR34","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/978-3-030-58517-4_21","volume-title":"Computer Vision - ECCV 2020","author":"H Kwon","year":"2020","unstructured":"Kwon, H., Kim, M., Kwak, S., Cho, M.: Motionsqueeze: Neural motion feature learning for video understanding. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision - ECCV 2020, pp. 345\u2013362. Springer International Publishing, Cham (2020)"},{"key":"2465_CR35","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: Tsm: temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"2465_CR36","doi-asserted-by":"crossref","unstructured":"Li, Y., Ji, B., Shi, X., Zhang, J., Kang, B., Wang, L.: Tea: temporal excitation and aggregation for action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 909\u2013918 (2020)","DOI":"10.1109\/CVPR42600.2020.00099"},{"key":"2465_CR37","doi-asserted-by":"publisher","first-page":"2453","DOI":"10.1007\/s11263-022-01661-1","volume":"130","author":"Y Tian","year":"2022","unstructured":"Tian, Y., Yan, Y., Zhai, G., Guo, G., Gao, Z.: Ean: event adaptive network for enhanced action recognition. Int. J. Comput. Vision 130, 2453\u20132471 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"2465_CR38","unstructured":"Huang, Z., Zhang, S., Pan, L., Qing, Z., Tang, M., Liu, Z., Ang\u00a0Jr, M.H.: Tada! temporally-adaptive convolutions for video understanding. In: ICLR (2022)"},{"key":"2465_CR39","unstructured":"Huang, Z., Zhang, S., Pan, L., Qing, Z., Zhang, Y., Liu, Z., Ang\u00a0Jr, M.H.: Temporally-adaptive models for efficient video understanding. arXiv preprint (2023). arXiv:2308.05787"},{"key":"2465_CR40","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, L., Wu, W., Qian, C., Lu, T.: Tam: temporal adaptive module for video recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 13708\u201313718 (2021)","DOI":"10.1109\/ICCV48922.2021.01345"},{"key":"2465_CR41","doi-asserted-by":"publisher","first-page":"3347","DOI":"10.1109\/TPAMI.2022.3173658","volume":"45","author":"M Wang","year":"2023","unstructured":"Wang, M., Xing, J., Su, J., Chen, J., Liu, Y.: Learning spatiotemporal and motion features in a unified 2d network for action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45, 3347\u20133362 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2022.3173658","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2465_CR42","doi-asserted-by":"crossref","unstructured":"Jiang, B., Wang, M., Gan, W., Wu, W., Yan, J.: Stm: spatiotemporal and motion encoding for action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2000\u20132009 (2019)","DOI":"10.1109\/ICCV.2019.00209"},{"key":"2465_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Z., She, Q., Smolic, A.: Action-net: multipath excitation for action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 13214\u201313223 (2021)","DOI":"10.1109\/CVPR46437.2021.01301"},{"key":"2465_CR44","doi-asserted-by":"publisher","first-page":"2990","DOI":"10.1109\/TMM.2020.2965434","volume":"22","author":"J Li","year":"2020","unstructured":"Li, J., Liu, X., Zhang, W., Zhang, M., Song, J., Sebe, N.: Spatio-temporal attention networks for action recognition and detection. IEEE Trans. Multimedia 22, 2990\u20133001 (2020). https:\/\/doi.org\/10.1109\/TMM.2020.2965434","journal-title":"IEEE Trans. Multimedia"},{"key":"2465_CR45","doi-asserted-by":"publisher","first-page":"416","DOI":"10.1109\/TMM.2018.2862341","volume":"21","author":"D Li","year":"2019","unstructured":"Li, D., Yao, T., Duan, L.-Y., Mei, T., Rui, Y.: Unified spatio-temporal attention networks for action recognition in videos. IEEE Trans. Multimedia 21, 416\u2013428 (2019). https:\/\/doi.org\/10.1109\/TMM.2018.2862341","journal-title":"IEEE Trans. Multimedia"},{"key":"2465_CR46","doi-asserted-by":"publisher","first-page":"5174","DOI":"10.1109\/TCSVT.2023.3250646","volume":"33","author":"Z Li","year":"2023","unstructured":"Li, Z., Li, J., Ma, Y., Wang, R., Shi, Z., Ding, Y., Liu, X.: Spatio-temporal adaptive network with bidirectional temporal difference for action recognition. IEEE Trans. Circuits Syst. Video Technol. 33, 5174\u20135185 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2023.3250646","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2465_CR47","doi-asserted-by":"publisher","first-page":"7594","DOI":"10.1109\/TMM.2022.3224327","volume":"25","author":"Z Xie","year":"2023","unstructured":"Xie, Z., Chen, J., Wu, K., Guo, D., Hong, R.: Global temporal difference network for action recognition. IEEE Trans. Multimedia 25, 7594\u20137606 (2023). https:\/\/doi.org\/10.1109\/TMM.2022.3224327","journal-title":"IEEE Trans. Multimedia"},{"key":"2465_CR48","doi-asserted-by":"publisher","first-page":"977","DOI":"10.1109\/TCSVT.2022.3207518","volume":"33","author":"X Sheng","year":"2023","unstructured":"Sheng, X., Li, K., Shen, Z., Xiao, G.: A progressive difference method for capturing visual tempos on action recognition. IEEE Trans. Circuits Syst. Video Technol. 33, 977\u2013987 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2022.3207518","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2465_CR49","doi-asserted-by":"publisher","first-page":"3912","DOI":"10.1109\/TCSVT.2023.3235522","volume":"33","author":"Y Chen","year":"2023","unstructured":"Chen, Y., Ge, H., Liu, Y., Cai, X., Sun, L.: Agpn: action granularity pyramid network for video action recognition. IEEE Trans. Circuits Syst. Video Technol. 33, 3912\u20133923 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2023.3235522","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2465_CR50","doi-asserted-by":"publisher","unstructured":"Wang, Q., Hu, Q., Gao, Z., Li, P., Hu, Q.: Ams-net: modeling adaptive multi-granularity spatio-temporal cues for video action recognition. IEEE Trans. Neural Netw. Learn. Syst. 1\u201315 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2023.3321141","DOI":"10.1109\/TNNLS.2023.3321141"},{"key":"2465_CR51","doi-asserted-by":"crossref","unstructured":"Neimark, D., Bar, O., Zohar, M., Asselmann, D.: Video transformer network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) Workshops, pp. 3163\u20133172 (2021)","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"2465_CR52","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding?. In: ICML vol.\u00a02, p.\u00a04 (2021)"},{"key":"2465_CR53","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"2465_CR54","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C.-Y., Fan, H., Mangalam, K., Xiong, B., Malik, J., Feichtenhofer, C.: Mvitv2: improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"issue":"10","key":"2465_CR55","doi-asserted-by":"publisher","first-page":"12581","DOI":"10.1109\/TPAMI.2023.3282631","volume":"45","author":"K Li","year":"2023","unstructured":"Li, K., Wang, Y., Zhang, J., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: unifying convolution and self-attention for visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(10), 12581\u201312600 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2465_CR56","doi-asserted-by":"crossref","unstructured":"Li, K., Li, X., Wang, Y., He, Y., Wang, Y., Wang, L., Qiao, Y.: Videomamba: state space model for efficient video understanding (2024). arXiv:2403.06977","DOI":"10.1007\/978-3-031-73347-5_14"},{"key":"2465_CR57","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K., B.: Pre-training of deep bidirectional transformers for language understanding, (2018). arXiv:1810.04805 arXiv preprint"},{"key":"2465_CR58","unstructured":"Chen, M., Radford, A., Child, R., Wu, J., Jun, H., Luan, D., Sutskever, I.: Generative pretraining from pixels. iIn: III, H.D., Singh, A.: (Eds.), Proceedings of the 37th International Conference on Machine Learning, volume 119 of Proceedings of Machine Learning Research, PMLR, pp. 1691\u20131703 (2020). https:\/\/proceedings.mlr.press\/v119\/chen20s.html"},{"key":"2465_CR59","doi-asserted-by":"publisher","first-page":"218","DOI":"10.1109\/TMM.2023.3263288","volume":"26","author":"Z Qing","year":"2024","unstructured":"Qing, Z., Zhang, S., Huang, Z., Wang, X., Wang, Y., Lv, Y., Gao, C., Sang, N.: Mar: masked autoencoders for efficient action recognition. IEEE Trans. Multimedia 26, 218\u2013233 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3263288","journal-title":"IEEE Trans. Multimedia"},{"key":"2465_CR60","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2465_CR61","doi-asserted-by":"publisher","unstructured":"Zhang, Y., Bai, Y., Liu, C., Wang, H., Li, S., Fu, Y.: Frame flexible network. In. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2023, 10504\u201310513 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01012","DOI":"10.1109\/CVPR52729.2023.01012"},{"key":"2465_CR62","doi-asserted-by":"publisher","first-page":"5458","DOI":"10.1109\/TMM.2022.3193057","volume":"25","author":"B Wang","year":"2023","unstructured":"Wang, B., Liu, C., Chang, F., Wang, W., Li, N.: Ae-net:adjoint enhancement network for efficient action recognition in video understanding. IEEE Trans. Multimedia 25, 5458\u20135468 (2023). https:\/\/doi.org\/10.1109\/TMM.2022.3193057","journal-title":"IEEE Trans. Multimedia"},{"key":"2465_CR63","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.124917","volume":"255","author":"L Li","year":"2024","unstructured":"Li, L., Tang, M., Yang, Z., Hu, J., Zhao, M.: Spatio-temporal adaptive convolution and bidirectional motion difference fusion for video action recognition. Expert Syst. Appl. 255, 124917 (2024). https:\/\/doi.org\/10.1016\/j.eswa.2024.124917. (https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0957417424017846)","journal-title":"Expert Syst. Appl."},{"key":"2465_CR64","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.113547","volume":"318","author":"L Li","year":"2025","unstructured":"Li, L., Tang, M., Qing, S., Zheng, Y., Hu, J., Zhao, M., Chen, S.: Action-prompt: a unified visual prompt and fusion network for enhanced video action recognition. Knowl.-Based Syst. 318, 113547 (2025). https:\/\/doi.org\/10.1016\/j.knosys.2025.113547. (https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0950705125005933)","journal-title":"Knowl.-Based Syst."},{"key":"2465_CR65","doi-asserted-by":"crossref","unstructured":"Wang, X., Gupta, A.: Videos as space-time region graphs. In: Proceedings of the European Conference on Computer Vision (ECCV) (2018)","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"2465_CR66","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Sun, X., Luo, C., Zha, Z.-J., Zeng, W.: Spatiotemporal fusion in 3d cnns: a probabilistic view. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00985"},{"key":"2465_CR67","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Li, X., Liu, C., Shuai, B., Zhu, Y., Brattoli B., Chen, H., Marsic, I., Tighe, J.: Vidtr: video transformer without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13577\u201313587 (2021)","DOI":"10.1109\/ICCV48922.2021.01332"},{"key":"2465_CR68","unstructured":"Bulat, A., Perez\u00a0Rua, J.M., Sudhakaran, S., Martinez, B., Tzimiropoulos, G.: Space-time mixing attention for video transformer. In: Ranzato, M., Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan J.W.: (Eds.), Advances in Neural Information Processing Systems, volume\u00a034, Curran Associates, Inc., pp. 19594\u201319607 (2021). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2021\/file\/a34bacf839b923770b2c360eefa26748-Paper.pdf"},{"key":"2465_CR69","doi-asserted-by":"publisher","first-page":"388","DOI":"10.1007\/978-3-031-19833-5_23","volume-title":"Computer Vision - ECCV 2022","author":"Z Lin","year":"2022","unstructured":"Lin, Z., Geng, S., Zhang, R., Gao, P., de Melo, G., Wang, X., Dai, J., Qiao, Y., Li, H.: Frozen clip models are efficient video learners. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision - ECCV 2022, pp. 388\u2013404. Springer Nature Switzerland, Cham (2022)"},{"key":"2465_CR70","doi-asserted-by":"publisher","first-page":"2496","DOI":"10.1109\/TNNLS.2022.3190367","volume":"35","author":"S Alfasly","year":"2024","unstructured":"Alfasly, S., Chui, C.K., Jiang, Q., Lu, J., Xu, C.: An effective video transformer with synchronized spatiotemporal and spatial self-attention for action recognition. IEEE Trans. Neural Netw. Learn. Syst. 35, 2496\u20132509 (2024). https:\/\/doi.org\/10.1109\/TNNLS.2022.3190367","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"2465_CR71","doi-asserted-by":"crossref","unstructured":"Yang, J., Dong, X., Liu, L., Zhang, C., Shen, J., Yu, D.: Recurring the transformer for video action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 14063\u201314073 (2022)","DOI":"10.1109\/CVPR52688.2022.01367"},{"key":"2465_CR72","doi-asserted-by":"publisher","unstructured":"Zhang, H., Hao, Y., Ngo, C.-W.: Token shift transformer for video classification. In: Proceedings of the 29th ACM International Conference on Multimedia, MM \u201921, Association for Computing Machinery, New York, NY, USA, p. 917\u2013925 (2021). https:\/\/doi.org\/10.1145\/3474085.3475272","DOI":"10.1145\/3474085.3475272"},{"key":"2465_CR73","unstructured":"Tan, H., Lei, J., Wolf, T., Bansal, M.: Vimpac: video pre-training via masked token prediction and contrastive learning, (2021). arXiv:2106.11250"},{"key":"2465_CR74","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., Torralba, A.: Learning deep features for discriminative localization. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2921\u20132929 (2016)","DOI":"10.1109\/CVPR.2016.319"},{"key":"2465_CR75","unstructured":"Van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-sne., J. Mach. Learn. Res. 9 (2008)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02465-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02465-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02465-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:30:21Z","timestamp":1784892621000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02465-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":75,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2465"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02465-w","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"22 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"402"}}