{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T16:36:57Z","timestamp":1775666217673,"version":"3.50.1"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T00:00:00Z","timestamp":1737417600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T00:00:00Z","timestamp":1737417600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.61673396"],"award-info":[{"award-number":["No.61673396"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["No.ZR2022MF260"],"award-info":[{"award-number":["No.ZR2022MF260"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-06927-2","type":"journal-article","created":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T02:54:15Z","timestamp":1737428055000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An efficient video transformer network with token discard and keyframe enhancement for action recognition"],"prefix":"10.1007","volume":"81","author":[{"given":"Qian","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zuosui","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingwen","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,21]]},"reference":[{"key":"6927_CR1","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. Advances in neural information processing systems 27"},{"key":"6927_CR2","doi-asserted-by":"crossref","unstructured":"Karpathy A, Toderici G, Shetty S, Leung T, Sukthankar R, Fei-Fei L (2014) Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1725\u20131732","DOI":"10.1109\/CVPR.2014.223"},{"key":"6927_CR3","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Pinz A, Zisserman A (2016) Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1933\u20131941","DOI":"10.1109\/CVPR.2016.213"},{"key":"6927_CR4","doi-asserted-by":"crossref","unstructured":"Wang L, Xiong Y, Wang Z, Qiao Y, Lin D, Tang X, Van\u00a0Gool L (2016) Temporal segment networks: Towards good practices for deep action recognition. In: European Conference on Computer Vision, pp. 20\u201336. Springer","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"6927_CR5","first-page":"3468","volume":"2","author":"R Christoph","year":"2016","unstructured":"Christoph R, Pinz FA (2016) Spatiotemporal residual networks for video action recognition. Advanc Neural Inf Process Syst 2:3468\u20133476","journal-title":"Advanc Neural Inf Process Syst"},{"key":"6927_CR6","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Fan H, Malik J, He K (2019) Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211","DOI":"10.1109\/ICCV.2019.00630"},{"issue":"1","key":"6927_CR7","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2012","unstructured":"Ji S, Xu W, Yang M, Yu K (2012) 3d convolutional neural networks for human action recognition. IEEE Trans Pattern Anal Mach Intell 35(1):221\u2013231","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6927_CR8","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"6927_CR9","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"6927_CR10","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C (2020) X3d: Expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"6927_CR11","unstructured":"Vaswani A (2017) Attention is all you need. Advances in Neural Information Processing Systems"},{"key":"6927_CR12","unstructured":"Devlin J (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"6927_CR13","unstructured":"Dosovitskiy A (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"6927_CR14","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, Sun C, Lu\u010di\u0107 M, Schmid C (2021) Vivit: A video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"6927_CR15","unstructured":"Bertasius G, Wang H, Torresani L (2021) Is space-time attention all you need for video understanding? In: ICML, 2 p. 4"},{"key":"6927_CR16","doi-asserted-by":"crossref","unstructured":"Park SH, Tack J, Heo B, Ha J-W, Shin J (2022) K-centered patch sampling for efficient video recognition. In: European Conference on Computer Vision, pp. 160\u2013176 . Springer","DOI":"10.1007\/978-3-031-19833-5_10"},{"key":"6927_CR17","doi-asserted-by":"crossref","unstructured":"Wang J, Yang X, Li H, Liu L, Wu Z, Jiang Y-G (2022) Efficient video transformers with spatial-temporal token selection. In: European Conference on Computer Vision, pp. 69\u201386. Springer","DOI":"10.1007\/978-3-031-19833-5_5"},{"key":"6927_CR18","doi-asserted-by":"crossref","unstructured":"Wang L, Huang B, Zhao Z, Tong Z, He Y, Wang Y, Wang Y, Qiao Y (2023) Videomae v2: Scaling video masked autoencoders with dual masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14549\u201314560","DOI":"10.1109\/CVPR52729.2023.01398"},{"issue":"4","key":"6927_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3633781","volume":"20","author":"Z Feng","year":"2024","unstructured":"Feng Z, Xu J, Ma L, Zhang S (2024) Efficient video transformers via spatial-temporal token merging for action recognition. ACM Trans Multimed Comput, Commun Appl 20(4):1\u201321","journal-title":"ACM Trans Multimed Comput, Commun Appl"},{"key":"6927_CR20","doi-asserted-by":"crossref","unstructured":"Wu Q, Cui R, Li Y, Zhu H (2024) Haltingvt: Adaptive token halting transformer for efficient video recognition. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4305\u20134309. IEEE","DOI":"10.1109\/ICASSP48485.2024.10447548"},{"key":"6927_CR21","doi-asserted-by":"crossref","unstructured":"Choi J, Lee S, Chu J, Choi M, Kim HJ (2024) vid-tldr: Training free token merging for light-weight video transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18771\u201318781","DOI":"10.1109\/CVPR52733.2024.01776"},{"key":"6927_CR22","doi-asserted-by":"publisher","first-page":"218","DOI":"10.1109\/TMM.2023.3263288","volume":"26","author":"Z Qing","year":"2023","unstructured":"Qing Z, Zhang S, Huang Z, Wang X, Wang Y, Lv Y, Gao C, Sang N (2023) Mar: masked autoencoders for efficient action recognition. IEEE Trans Multimed 26:218\u2013233","journal-title":"IEEE Trans Multimed"},{"key":"6927_CR23","doi-asserted-by":"crossref","unstructured":"Fan H, Xiong B, Mangalam K, Li Y, Yan Z, Malik J, Feichtenhofer C (2021) Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6824\u20136835","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"6927_CR24","doi-asserted-by":"crossref","unstructured":"Neimark D, Bar O, Zohar M, Asselmann D (2021) Video transformer network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3163\u20133172","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"6927_CR25","first-page":"12493","volume":"34","author":"M Patrick","year":"2021","unstructured":"Patrick M, Campbell D, Asano Y, Misra I, Metze F, Feichtenhofer C, Vedaldi A, Henriques JF (2021) Keeping your eye on the ball: trajectory attention in video transformers. Advan Neural Inf Process Syst 34:12493\u201312506","journal-title":"Advan Neural Inf Process Syst"},{"key":"6927_CR26","doi-asserted-by":"crossref","unstructured":"Liu Z, Ning J, Cao Y, Wei Y, Zhang Z, Lin S, Hu H (2022) Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"6927_CR27","doi-asserted-by":"crossref","unstructured":"Truong T-D, Bui Q-H, Duong CN, Seo H-S, Phung SL, Li X, Luu K (2022) Direcformer: A directed attention in transformer approach to robust action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20030\u201320040","DOI":"10.1109\/CVPR52688.2022.01940"},{"key":"6927_CR28","unstructured":"Li K, Wang Y, Gao P, Song G, Liu Y, Li H, Qiao Y (2022) Uniformer: unified transformer for efficient spatiotemporal representation learning. arXiv preprint arXiv:2201.04676"},{"key":"6927_CR29","doi-asserted-by":"crossref","unstructured":"Yan S, Xiong X, Arnab A, Lu Z, Zhang M, Sun C, Schmid C (2022) Multiview transformers for video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3333\u20133343","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"6927_CR30","doi-asserted-by":"crossref","unstructured":"Herzig R, Ben-Avraham E, Mangalam K, Bar A, Chechik G, Rohrbach A, Darrell T, Globerson A (2022) Object-region video transformers. In: Proceedings of the Ieee\/cvf Conference on Computer Vision and Pattern Recognition, pp. 3148\u20133159","DOI":"10.1109\/CVPR52688.2022.00315"},{"key":"6927_CR31","first-page":"13937","volume":"34","author":"Y Rao","year":"2021","unstructured":"Rao Y, Zhao W, Liu B, Lu J, Zhou J, Hsieh C-J (2021) Dynamicvit: efficient vision transformers with dynamic token sparsification. Advan Neural Inf Process Syst 34:13937\u201313949","journal-title":"Advan Neural Inf Process Syst"},{"key":"6927_CR32","unstructured":"Liang Y, Ge C, Tong Z, Song Y, Wang J, Xie P (2022) Not all patches are what you need: Expediting vision transformers via token reorganizations. arXiv preprint arXiv:2202.07800"},{"key":"6927_CR33","doi-asserted-by":"crossref","unstructured":"Fayyaz M, Koohpayegani SA, Jafari FR, Sengupta S, Joze HRV, Sommerlade E, Pirsiavash H, Gall J (2022) Adaptive token sampling for efficient vision transformers. In: European Conference on Computer Vision, pp. 396\u2013414. Springer","DOI":"10.1007\/978-3-031-20083-0_24"},{"key":"6927_CR34","doi-asserted-by":"crossref","unstructured":"Kong Z, Dong P, Ma X, Meng X, Niu W, Sun M, Shen X, Yuan G, Ren B, Tang H et al (2022) Spvit: Enabling faster vision transformers via latency-aware soft token pruning. In: European Conference on Computer Vision, pp. 620\u2013640. Springer","DOI":"10.1007\/978-3-031-20083-0_37"},{"key":"6927_CR35","doi-asserted-by":"crossref","unstructured":"Yin H, Vahdat A, Alvarez JM, Mallya A, Kautz J, Molchanov P (2022) A-vit: Adaptive tokens for efficient vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10809\u201310818","DOI":"10.1109\/CVPR52688.2022.01054"},{"key":"6927_CR36","doi-asserted-by":"crossref","unstructured":"Meng L, Li H, Chen B-C, Lan S, Wu Z, Jiang Y-G, Lim S-N (2022) Adavit: Adaptive vision transformers for efficient image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12309\u201312318","DOI":"10.1109\/CVPR52688.2022.01199"},{"key":"6927_CR37","unstructured":"Bolya D, Fu C-Y, Dai X, Zhang P, Feichtenhofer C, Hoffman J (2022) Token merging: Your vit but faster. arXiv preprint arXiv:2210.09461"},{"key":"6927_CR38","doi-asserted-by":"crossref","unstructured":"Feng Z, Zhang S (2023) Efficient vision transformer via token merger. IEEE Transactions on Image Processing","DOI":"10.1109\/TIP.2023.3293763"},{"key":"6927_CR39","doi-asserted-by":"crossref","unstructured":"Chen L, Tong Z, Song Y, Wu G, Wang L (2023) Efficient video action detection with token dropout and context refinement. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10388\u201310399","DOI":"10.1109\/ICCV51070.2023.00953"},{"key":"6927_CR40","doi-asserted-by":"crossref","unstructured":"Chen J, Ho CM (2022) Mm-vit: Multi-modal video transformer for compressed video action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1910\u20131921","DOI":"10.1109\/WACV51458.2022.00086"},{"key":"6927_CR41","doi-asserted-by":"crossref","unstructured":"Goyal R, Ebrahimi\u00a0Kahou S, Michalski V, Materzynska J, Westphal S, Kim H, Haenel V, Fruend I, Yianilos P, Mueller-Freitag M et al (2017) The\" something something\" video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850","DOI":"10.1109\/ICCV.2017.622"},{"key":"6927_CR42","doi-asserted-by":"crossref","unstructured":"Lin J, Gan C, Han S (2019) Tsm: Temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7083\u20137093","DOI":"10.1109\/ICCV.2019.00718"},{"key":"6927_CR43","unstructured":"Fan Q, Chen C-FR, Kuehne H, Pistoia M, Cox D (2019) More is less: Learning efficient video representations by big-little network and depthwise temporal aggregation. Advances in Neural Information Processing Systems. 32"},{"key":"6927_CR44","unstructured":"Fan Q, Panda R et al (2021) An image classifier can suffice for video understanding. arXiv preprint arXiv:2106.141042"},{"key":"6927_CR45","doi-asserted-by":"crossref","unstructured":"Jiang B, Wang M, Gan W, Wu W, Yan J (2019) Stm: spatiotemporal and motion encoding for action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2000\u20132009","DOI":"10.1109\/ICCV.2019.00209"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06927-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-06927-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06927-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T02:54:51Z","timestamp":1737428091000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-06927-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,21]]},"references-count":45,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025,1]]}},"alternative-id":["6927"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-06927-2","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,21]]},"assertion":[{"value":"7 January 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 January 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The materials are available from the corresponding author.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Materials availability"}}],"article-number":"408"}}