{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T00:26:57Z","timestamp":1759796817254,"version":"build-2065373602"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T00:00:00Z","timestamp":1756857600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T00:00:00Z","timestamp":1756857600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.61673396"],"award-info":[{"award-number":["No.61673396"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["No.ZR2022MF260"],"award-info":[{"award-number":["No.ZR2022MF260"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s10586-025-05478-8","type":"journal-article","created":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T14:08:53Z","timestamp":1756908533000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Midframe-centric token merging for efficient video transformer"],"prefix":"10.1007","volume":"28","author":[{"given":"Qian","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zuosui","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingwen","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,3]]},"reference":[{"key":"5478_CR1","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. Advances in neural information processing systems 27 (2014)"},{"key":"5478_CR2","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1933\u20131941 (2016)","DOI":"10.1109\/CVPR.2016.213"},{"key":"5478_CR3","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Toderici, G., Shetty, S., Leung, T., Sukthankar, R., Fei-Fei, L.: Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1725\u20131732 (2014)","DOI":"10.1109\/CVPR.2014.223"},{"key":"5478_CR4","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Van\u00a0Gool, L.: Temporal segment networks: Towards good practices for deep action recognition. In: European Conference on Computer Vision, pp. 20\u201336 (2016). Springer","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"5478_CR5","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"5478_CR6","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3d: Expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"5478_CR7","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"5478_CR8","first-page":"3468","volume":"2","author":"R Christoph","year":"2016","unstructured":"Christoph, R., Pinz, F.A.: Spatiotemporal residual networks for video action recognition. Adv. Neural Inf. Proc. Syst. 2, 3468\u20133476 (2016)","journal-title":"Adv. Neural Inf. Proc. Syst."},{"key":"5478_CR9","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"5478_CR10","unstructured":"Dosovitskiy, A.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"5478_CR11","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, pp. 10347\u201310357 (2021). PMLR"},{"key":"5478_CR12","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: A video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"5478_CR13","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, vol. 2, p. 4 (2021)"},{"key":"5478_CR14","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, X., Arnab, A., Lu, Z., Zhang, M., Sun, C., Schmid, C.: Multiview transformers for video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3333\u20133343 (2022)","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"5478_CR15","first-page":"12493","volume":"34","author":"M Patrick","year":"2021","unstructured":"Patrick, M., Campbell, D., Asano, Y., Misra, I., Metze, F., Feichtenhofer, C., Vedaldi, A., Henriques, J.F.: Keeping your eye on the ball: Trajectory attention in video transformers. Adv. Neural Inf. Proc. Syst. 34, 12493\u201312506 (2021)","journal-title":"Adv. Neural Inf. Proc. Syst."},{"key":"5478_CR16","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., Feichtenhofer, C.: Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6824\u20136835 (2021)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"5478_CR17","doi-asserted-by":"crossref","unstructured":"Neimark, D., Bar, O., Zohar, M., Asselmann, D.: Video transformer network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3163\u20133172 (2021)","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"5478_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., Hu, H.: Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211 (2022)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"5478_CR19","unstructured":"Li, K., Wang, Y., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: Unified transformer for efficient spatiotemporal representation learning. arXiv preprint arXiv:2201.04676 (2022)"},{"key":"5478_CR20","doi-asserted-by":"crossref","unstructured":"Truong, T.-D., Bui, Q.-H., Duong, C.N., Seo, H.-S., Phung, S.L., Li, X., Luu, K.: Direcformer: A directed attention in transformer approach to robust action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20030\u201320040 (2022)","DOI":"10.1109\/CVPR52688.2022.01940"},{"key":"5478_CR21","first-page":"13937","volume":"34","author":"Y Rao","year":"2021","unstructured":"Rao, Y., Zhao, W., Liu, B., Lu, J., Zhou, J., Hsieh, C.-J.: Dynamicvit: Efficient vision transformers with dynamic token sparsification. Adv. Neural Inf. Proc. Syst. 34, 13937\u201313949 (2021)","journal-title":"Adv. Neural Inf. Proc. Syst."},{"key":"5478_CR22","unstructured":"Liang, Y., Ge, C., Tong, Z., Song, Y., Wang, J., Xie, P.: Not all patches are what you need: Expediting vision transformers via token reorganizations. arXiv preprint arXiv:2202.07800 (2022)"},{"key":"5478_CR23","doi-asserted-by":"crossref","unstructured":"Kong, Z., Dong, P., Ma, X., Meng, X., Niu, W., Sun, M., Shen, X., Yuan, G., Ren, B., Tang, H., et al.: Spvit: Enabling faster vision transformers via latency-aware soft token pruning. In: European Conference on Computer Vision, pp. 620\u2013640 (2022). Springer","DOI":"10.1007\/978-3-031-20083-0_37"},{"key":"5478_CR24","doi-asserted-by":"crossref","unstructured":"Yin, H., Vahdat, A., Alvarez, J.M., Mallya, A., Kautz, J., Molchanov, P.: A-vit: Adaptive tokens for efficient vision transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10809\u201310818 (2022)","DOI":"10.1109\/CVPR52688.2022.01054"},{"key":"5478_CR25","doi-asserted-by":"crossref","unstructured":"Meng, L., Li, H., Chen, B.-C., Lan, S., Wu, Z., Jiang, Y.-G., Lim, S.-N.: Adavit: Adaptive vision transformers for efficient image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12309\u201312318 (2022)","DOI":"10.1109\/CVPR52688.2022.01199"},{"key":"5478_CR26","doi-asserted-by":"crossref","unstructured":"Fayyaz, M., Koohpayegani, S.A., Jafari, F.R., Sengupta, S., Joze, H.R.V., Sommerlade, E., Pirsiavash, H., Gall, J.: Adaptive token sampling for efficient vision transformers. In: European Conference on Computer Vision, pp. 396\u2013414 (2022). Springer","DOI":"10.1007\/978-3-031-20083-0_24"},{"key":"5478_CR27","unstructured":"Bolya, D., Fu, C.-Y., Dai, X., Zhang, P., Feichtenhofer, C., Hoffman, J.: Token merging: Your vit but faster. arXiv preprint arXiv:2210.09461 (2022)"},{"key":"5478_CR28","doi-asserted-by":"publisher","first-page":"4156","DOI":"10.1109\/TIP.2023.3293763","volume":"32","author":"Z Feng","year":"2023","unstructured":"Feng, Z., Zhang, S.: Efficient vision transformer via token merger. IEEE Trans. Image Proc. 32, 4156\u20134169 (2023)","journal-title":"IEEE Trans. Image Proc."},{"key":"5478_CR29","doi-asserted-by":"crossref","unstructured":"Park, S.H., Tack, J., Heo, B., Ha, J.-W., Shin, J.: K-centered patch sampling for efficient video recognition. In: European Conference on Computer Vision, pp. 160\u2013176 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_10"},{"key":"5478_CR30","doi-asserted-by":"crossref","unstructured":"Wang, J., Yang, X., Li, H., Liu, L., Wu, Z., Jiang, Y.-G.: Efficient video transformers with spatial-temporal token selection. In: European Conference on Computer Vision, pp. 69\u201386 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_5"},{"issue":"4","key":"5478_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3633781","volume":"20","author":"Z Feng","year":"2024","unstructured":"Feng, Z., Xu, J., Ma, L., Zhang, S.: Efficient video transformers via spatial-temporal token merging for action recognition. ACM Trans. Multimedia Comput., Commun. Appl. 20(4), 1\u201321 (2024)","journal-title":"ACM Trans. Multimedia Comput., Commun. Appl."},{"key":"5478_CR32","doi-asserted-by":"crossref","unstructured":"Choi, J., Lee, S., Chu, J., Choi, M., Kim, H.J.: vid-tldr: Training free token merging for light-weight video transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18771\u201318781 (2024)","DOI":"10.1109\/CVPR52733.2024.01776"},{"key":"5478_CR33","doi-asserted-by":"crossref","unstructured":"Wu, Q., Cui, R., Li, Y., Zhu, H.: Haltingvt: Adaptive token halting transformer for efficient video recognition. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4305\u20134309 (2024). IEEE","DOI":"10.1109\/ICASSP48485.2024.10447548"},{"key":"5478_CR34","unstructured":"Devlin, J.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"5478_CR35","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229 (2020). Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"5478_CR36","first-page":"79399","volume":"36","author":"D Lee","year":"2024","unstructured":"Lee, D., Lee, J., Choi, J.: Cast: cross-attention in space and time for video action recognition. Adv. Neural Inf. Proc. Syst. 36, 79399\u201379425 (2024)","journal-title":"Adv. Neural Inf. Proc. Syst."},{"key":"5478_CR37","doi-asserted-by":"crossref","unstructured":"Korbar, B., Tran, D., Torresani, L.: Scsampler: Sampling salient clips from video for efficient action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6232\u20136242 (2019)","DOI":"10.1109\/ICCV.2019.00633"},{"issue":"4","key":"5478_CR38","doi-asserted-by":"publisher","first-page":"1699","DOI":"10.1109\/TPAMI.2020.3029425","volume":"44","author":"Z Wu","year":"2020","unstructured":"Wu, Z., Li, H., Xiong, C., Jiang, Y.-G., Davis, L.S.: A dynamic frame selection framework for fast video recognition. IEEE Trans. Pattern Anal. Mach. Intell. 44(4), 1699\u20131711 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5478_CR39","doi-asserted-by":"crossref","unstructured":"Bhardwaj, S., Srinivasan, M., Khapra, M.M.: Efficient video classification using fewer frames. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 354\u2013363 (2019)","DOI":"10.1109\/CVPR.2019.00044"},{"issue":"3","key":"5478_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3571735","volume":"19","author":"H Tang","year":"2023","unstructured":"Tang, H., Ding, L., Wu, S., Ren, B., Sebe, N., Rota, P.: Deep unsupervised key frame extraction for efficient video classification. ACM Trans. Multimedia Comput. Commun. Appl. 19(3), 1\u201317 (2023)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"key":"5478_CR41","doi-asserted-by":"crossref","unstructured":"Gowda, S.N., Rohrbach, M., Sevilla-Lara, L.: Smart frame selection for action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 1451\u20131459 (2021)","DOI":"10.1609\/aaai.v35i2.16235"},{"key":"5478_CR42","doi-asserted-by":"crossref","unstructured":"Meng, Y., Lin, C.-C., Panda, R., Sattigeri, P., Karlinsky, L., Oliva, A., Saenko, K., Feris, R.: Ar-net: Adaptive frame resolution for efficient action recognition. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII 16, pp. 86\u2013104 (2020). Springer","DOI":"10.1007\/978-3-030-58571-6_6"},{"key":"5478_CR43","doi-asserted-by":"crossref","unstructured":"Lin, J., Duan, H., Chen, K., Lin, D., Wang, L.: Ocsampler: Compressing videos to one clip with single-step sampling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13894\u201313903 (2022)","DOI":"10.1109\/CVPR52688.2022.01352"},{"key":"5478_CR44","doi-asserted-by":"crossref","unstructured":"Ghodrati, A., Bejnordi, B.E., Habibian, A.: Frameexit: Conditional early exiting for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15608\u201315618 (2021)","DOI":"10.1109\/CVPR46437.2021.01535"},{"key":"5478_CR45","unstructured":"Raviv, A., Dinai, Y., Drozdov, I., Zehngut, N., Goldin, I., Center, S.I.R.: D-step: Dynamic spatio-temporal pruning. In: BMVC, p. 632 (2022)"},{"key":"5478_CR46","doi-asserted-by":"crossref","unstructured":"Sun, X., Panda, R., Chen, C.-F.R., Oliva, A., Feris, R., Saenko, K.: Dynamic network quantization for efficient video inference. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7375\u20137385 (2021)","DOI":"10.1109\/ICCV48922.2021.00728"},{"key":"5478_CR47","doi-asserted-by":"crossref","unstructured":"Pan, Z., Zhuang, B., He, H., Liu, J., Cai, J.: Less is more: Pay less attention in vision transformers. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 2035\u20132043 (2022)","DOI":"10.1609\/aaai.v36i2.20099"},{"key":"5478_CR48","unstructured":"Huang, H., Zhou, X., Cao, J., He, R., Tan, T.: Vision transformer with super token sampling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22690\u201322699 (2023)"},{"key":"5478_CR49","unstructured":"Yang, C., Xu, J., De\u00a0Mello, S., Crowley, E.J., Wang, X.: Gpvit: a high resolution non-hierarchical vision transformer with group propagation. arXiv preprint arXiv:2212.06795 (2022)"},{"key":"5478_CR50","unstructured":"Marin, D., Chang, J.-H.R., Ranjan, A., Prabhu, A., Rastegari, M., Tuzel, O.: Token pooling in vision transformers. arXiv preprint arXiv:2110.03860 (2021)"},{"key":"5478_CR51","doi-asserted-by":"crossref","unstructured":"Zeng, W., Jin, S., Liu, W., Qian, C., Luo, P., Ouyang, W., Wang, X.: Not all tokens are equal: Human-centric visual analysis via token clustering transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11101\u201311111 (2022)","DOI":"10.1109\/CVPR52688.2022.01082"},{"key":"5478_CR52","doi-asserted-by":"crossref","unstructured":"Long, S., Zhao, Z., Pi, J., Wang, S., Wang, J.: Beyond attentive tokens: Incorporating token importance and diversity for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10334\u201310343 (2023)","DOI":"10.1109\/CVPR52729.2023.00996"},{"key":"5478_CR53","unstructured":"Han, T., Xie, W., Zisserman, A.: Turbo training with token dropout. arXiv preprint arXiv:2210.04889 (2022)"},{"key":"5478_CR54","doi-asserted-by":"crossref","unstructured":"Goyal, R., Ebrahimi\u00a0Kahou, S., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., et al.: The\" something something\" video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"5478_CR55","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: Tsm: Temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7083\u20137093 (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"5478_CR56","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Luo, C., Tang, C., Chen, D., Codella, N., Zha, Z.-J.: Streaming video model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14602\u201314612 (2023)","DOI":"10.1109\/CVPR52729.2023.01403"},{"key":"5478_CR57","doi-asserted-by":"crossref","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., Shah, M.: Vita-clip: Video and text adaptive clip via multimodal prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23034\u201323044 (2023)","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"5478_CR58","doi-asserted-by":"crossref","unstructured":"Wang, M., Xing, J., Jiang, B., Chen, J., Mei, J., Zuo, X., Dai, G., Wang, J., Liu, Y.: M2-clip: A multimodal, multi-task adapting framework for video action recognition. arXiv preprint arXiv:2401.11649 (2024)","DOI":"10.1609\/aaai.v38i6.28361"},{"key":"5478_CR59","doi-asserted-by":"crossref","unstructured":"Long, F., Qiu, Z., Pan, Y., Yao, T., Luo, J., Mei, T.: Stand-alone inter-frame attention in video models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3192\u20133201 (2022)","DOI":"10.1109\/CVPR52688.2022.00319"},{"key":"5478_CR60","doi-asserted-by":"crossref","unstructured":"Long, F., Qiu, Z., Pan, Y., Yao, T., Ngo, C.-W., Mei, T.: Dynamic temporal filtering in video models. In: European Conference on Computer Vision, pp. 475\u2013492 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_28"},{"key":"5478_CR61","unstructured":"Li, K., Wang, Y., He, Y., Li, Y., Wang, Y., Wang, L., Qiao, Y.: Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer. arXiv preprint arXiv:2211.09552 (2022)"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-025-05478-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-025-05478-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-025-05478-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,6]],"date-time":"2025-10-06T09:35:17Z","timestamp":1759743317000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-025-05478-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,3]]},"references-count":61,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["5478"],"URL":"https:\/\/doi.org\/10.1007\/s10586-025-05478-8","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"type":"print","value":"1386-7857"},{"type":"electronic","value":"1573-7543"}],"subject":[],"published":{"date-parts":[[2025,9,3]]},"assertion":[{"value":"15 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 March 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The materials are available from the corresponding author.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Materials availability"}}],"article-number":"652"}}