{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T13:53:49Z","timestamp":1782395629626,"version":"3.54.5"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T00:00:00Z","timestamp":1779926400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T00:00:00Z","timestamp":1779926400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"BUPT Excellent Ph. D. Students Foundation","award":["CX20242004"],"award-info":[{"award-number":["CX20242004"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62502014"],"award-info":[{"award-number":["62502014"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Real-Time Image Proc"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11554-026-01893-1","type":"journal-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T14:56:32Z","timestamp":1779980192000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["HyBridge: hybrid decoupling-to-recoupling adaptation of vision-language models for real-time video action recognition"],"prefix":"10.1007","volume":"23","author":[{"given":"Mengyu","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ye","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lanshan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gongli","family":"Xi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wendong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,28]]},"reference":[{"key":"1893_CR1","doi-asserted-by":"crossref","unstructured":"Tian, Y., Yang, M., Zhang, L., Zhang, Z., Liu, Y., Xie, X., Que, X., Wang, W.: View while moving: Efficient video recognition in long-untrimmed videos. In Proceedings of the 31st ACM International Conference on Multimedia, pages 173\u2013183, (2023)","DOI":"10.1145\/3581783.3612035"},{"issue":"5","key":"1893_CR2","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1007\/s11554-024-01541-6","volume":"21","author":"S Yanxiong","year":"2024","unstructured":"Yanxiong, S., Zhao, Q.: Efficient spatio-temporal network for action recognition. J. Real-Time Image Proc. 21(5), 158 (2024)","journal-title":"J. Real-Time Image Proc."},{"issue":"2","key":"1893_CR3","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1007\/s11554-025-01662-6","volume":"22","author":"Q Zhao","year":"2025","unstructured":"Zhao, Q., Yanxiong, S., Zhang, H.: Stme-net: spatio-temporal motion excitation network for action recognition. J. Real-Time Image Proc. 22(2), 88 (2025)","journal-title":"J. Real-Time Image Proc."},{"key":"1893_CR4","doi-asserted-by":"crossref","unstructured":"Aravinda, C.V., Al-Shehari, T., Alsadhan, N.A., Shashank Shetty, G., Padmajadevi, KR Udaya Kumar Reddy.: A novel hybrid architecture for video frame prediction: combining convolutional lstm and 3d cnn. J. Real-Time Image Proc. 22(1), 50 (2025)","DOI":"10.1007\/s11554-025-01626-w"},{"issue":"2","key":"1893_CR5","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1007\/s11554-024-01435-7","volume":"21","author":"V Sharma","year":"2024","unstructured":"Sharma, V., Sharma, A., Saini, S.: Real-time attention-based embedded lstm for dynamic sign language recognition on edge devices. J. Real-Time Image Proc. 21(2), 53 (2024)","journal-title":"J. Real-Time Image Proc."},{"key":"1893_CR6","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In NIPS, pages 568\u2013576, (2014)"},{"key":"1893_CR7","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Gool, L.V.: Temporal segment networks: Towards good practices for deep action recognition. In ECCV, pages 20\u201336. Springer, (2016)","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1893_CR8","doi-asserted-by":"crossref","unstructured":"Donahue, J., Hendricks, L.A., Guadarrama, S., Rohrbach, M., Venugopalan, S., Saenko, K., Darrell, T.: Long-term recurrent convolutional networks for visual recognition and description. In CVPR (2625\u20132634), (2015)","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"1893_CR9","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In International conference on machine learning, pages 8748\u20138763. PmLR, (2021)"},{"key":"1893_CR10","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y-T., Parekh, Z., Pham, H., Le, Q., Sung, Y.-H., Li, Z., Duerig, T.: Scaling up visual and vision-language representation learning with noisy text supervision. In International conference on machine learning, pages 4904\u20134916. PMLR, (2021)"},{"key":"1893_CR11","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C., Hoi, S.C.H.: Align before fuse: Vision and language representation learning with momentum distillation. Adv. Neural. Inf. Process. Syst. 34, 9694\u20139705 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1893_CR12","unstructured":"Wang, M., Xing, J., Mei, J., Liu, Y., Jiang, Y.: Actionclip: Adapting language-image pretrained models for video action recognition. IEEE Transactions on Neural Networks and Learning Systems, (2023)"},{"key":"1893_CR13","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B-C., Cardie, C., Belongie, S., Hariharan, B., Lim, S-N.: Visual prompt tuning. In European conference on computer vision, pages 709\u2013727. Springer, (2022)","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"1893_CR14","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., Xie, W.: Prompting visual-language models for efficient video understanding. In European Conference on Computer Vision, pages 105\u2013124. Springer, (2022)","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"1893_CR15","unstructured":"Yang, T., Zhu, Y., Xie, Y., Zhang, A., Chen, C., Li, M.: Aim: Adapting image models for efficient video action recognition. arXiv preprint arXiv:2302.03024, (2023)"},{"key":"1893_CR16","doi-asserted-by":"crossref","unstructured":"Lin, Z., Geng, S., Zhang, R., Gao, P., Melo, G.D., Wang, X., Dai, J., Qiao, Y., Li, H.: Frozen clip models are efficient video learners. In European Conference on Computer Vision, pages 388\u2013404. Springer, (2022)","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"1893_CR17","doi-asserted-by":"publisher","first-page":"26462","DOI":"10.52202\/068431-1919","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., Li, H.: St-adapter: Parameter-efficient image-to-video transfer learning. Adv. Neural. Inf. Process. Syst. 35, 26462\u201326477 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1893_CR18","doi-asserted-by":"publisher","first-page":"16664","DOI":"10.52202\/068431-1212","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen, S., Ge, C., Tong, Z., Wang, J., Song, Y., Wang, J., Luo, P.: Adaptformer: Adapting vision transformers for scalable visual recognition. Adv. Neural. Inf. Process. Syst. 35, 16664\u201316678 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1893_CR19","doi-asserted-by":"crossref","unstructured":"Wu, W., Wang, X., Luo, H., Wang, J., Yang, Y., Ouyang, W.: Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 6620\u20136630, (2023)","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"1893_CR20","doi-asserted-by":"crossref","unstructured":"Ni, B., Peng, H., Chen, M., Zhang, S., Meng, G., Fu, J., Xiang, S., Ling, H.: Expanding language-image pretrained models for general video recognition. In European conference on computer vision, pages 1\u201318. Springer, (2022)","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"1893_CR21","doi-asserted-by":"crossref","unstructured":"Liu, R., Huang, J., Li, Feng, J., Wu, X., Li, T.H.: Revisiting temporal modeling for clip-based image-to-video knowledge transferring. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 6555\u20136564, (2023)","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"1893_CR22","doi-asserted-by":"crossref","unstructured":"Park, J., Lee, J., Sohn, K.: Dual-path adaptation from image to video transformers. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 2203\u20132213, (2023)","DOI":"10.1109\/CVPR52729.2023.00219"},{"key":"1893_CR23","doi-asserted-by":"crossref","unstructured":"Wang, Q., Du, J., Yan, K., Ding, S.: Seeing in flowing: Adapting clip for action recognition with motion prompts learning. In Proceedings of the 31st ACM International Conference on Multimedia, pages 5339\u20135347, (2023)","DOI":"10.1145\/3581783.3612490"},{"key":"1893_CR24","doi-asserted-by":"crossref","unstructured":"Saadi, I., Hadid, A., Cunningham, D.W., Taleb-Ahmed, A., El\u00a0Hillali, Y.: Pe-clip: A parameter-efficient fine-tuning of vision language models for dynamic facial expression recognition. ACM Transactions on Multimedia Computing, Communications and Applications, (2025)","DOI":"10.1145\/3786789"},{"key":"1893_CR25","doi-asserted-by":"publisher","first-page":"5517","DOI":"10.1609\/aaai.v38i6.28361","volume":"38","author":"M Wang","year":"2024","unstructured":"Wang, M., Xing, J., Jiang, B., Chen, J., Mei, J., Zuo, X., Dai, G., Wang, J., Liu, Y.: A multimodal, multi-task adapting framework for video action recognition. In Proceedings of the AAAI Conference on Artificial Intelligence 38, 5517\u20135525 (2024)","journal-title":"In Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1893_CR26","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C-J., Chang, K-W.: Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557, (2019)"},{"key":"1893_CR27","unstructured":"Jiasen, L., Batra, D., Parikh, D., Lee, S.: Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural. Inf. Process. Syst. 32, (2019)"},{"key":"1893_CR28","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141, Polosukhin, I.: Attention is all you need. NIPS 30, (2017)"},{"key":"1893_CR29","doi-asserted-by":"crossref","unstructured":"Dou, Z.-Y., Yichong, X., Gan, Z., Wang, J., Wang, S., Wang, L., Zhu, C., Pengchuan Zhang, L., Yuan, N.P., et al.: An empirical study of training end-to-end vision-and-language transformers. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 18166\u201318176 (2022)","DOI":"10.1109\/CVPR52688.2022.01763"},{"key":"1893_CR30","unstructured":"Zeng, Y., Zhang, X., Li, H.: Multi-grained vision language pre-training: Aligning texts with visual concepts. arXiv preprint arXiv:2111.08276, (2021)"},{"key":"1893_CR31","unstructured":"Bahng, H., Jahanian, A., Sankaranarayanan, S., Isola, P.: Exploring visual prompts for adapting large-scale models. arXiv preprint arXiv:2203.17274, (2022)"},{"key":"1893_CR32","doi-asserted-by":"crossref","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., Shah, M.: Vita-clip: Video and text adaptive clip via multimodal prompting. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 23034\u201323044 (2023)","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"1893_CR33","unstructured":"Will Kay, Joao Carreira, Karen Simonyan, Brian Zhang, Chloe Hillier, Sudheendra Vijayanarasimhan, Fabio Viola, Tim Green, Trevor Back, Paul Natsev, et\u00a0al. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950, 2017"},{"key":"1893_CR34","unstructured":"Wang, Y., He, Y., Li, Y., Li, K., Yu, J., Ma, X., Li, X., Chen, G., Chen, X., Wang, Y., et\u00a0al.: Internvid: A large-scale video-text dataset for multimodal understanding and generation. arXiv preprint arXiv:2307.06942, (2023)"},{"key":"1893_CR35","unstructured":"Yu, J., Wang, Z., Vasudevan, V., Yeung, L., Seyedhosseini, M., Wu, Y.: Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917, (2022)"},{"key":"1893_CR36","unstructured":"Wang, Y., Li, K., Li, Y., He, Y., Huang, B., Zhao, Z., Zhang, H., Xu, J., Liu, Y., Wang, Z., et\u00a0al.: Internvideo: General video foundation models via generative and discriminative learning. arXiv:2212.03191, (2022)"},{"key":"1893_CR37","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., Feichtenhofer, C.: Multiscale vision transformers. In ICCV, pages 6824\u20136835, (2021)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"1893_CR38","doi-asserted-by":"crossref","unstructured":"Li, Y., Chao-Yuan, W., Fan, H., Mangalam, K., Xiong, B., Malik, J., Feichtenhofer, C.: Mvitv 2: Improved multiscale vision transformers for classification and detection. In CVPR 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"1893_CR39","unstructured":"Li, K., Wang, Y., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: Unified transformer for efficient spatiotemporal representation learning. arXiv preprint arXiv:2201.04676, (2022)"},{"key":"1893_CR40","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In ICML (2), 4 (2021)"},{"key":"1893_CR41","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: A video vision transformer. In ICCV 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"1893_CR42","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., Han, H.: Video swin transformer. In CVPR 3202\u20133211 (2022)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"1893_CR43","doi-asserted-by":"crossref","unstructured":"Chen, T., Hongshan, Yu., Yang, Z., Li, Z., Sun, W., Chen, C.: Ost: Refining text knowledge with optimal spatio-temporal descriptor for general video recognition. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 18888\u201318898 (2024)","DOI":"10.1109\/CVPR52733.2024.01787"},{"key":"1893_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, B., Li, Y., Li, M., Zhang, J., Zhang, Q.: Cross attention guided multimodal network for video action recognition. In Chinese Conference on Pattern Recognition and Computer Vision (PRCV), pages 188\u2013202. Springer, (2025)","DOI":"10.1007\/978-981-95-5676-2_13"},{"key":"1893_CR45","doi-asserted-by":"crossref","unstructured":"Wang, W., Su, Y., Gu., J.: Transclip: Transferring vision\u2013language models for efficient video action recognition. IEEE Transactions on Industrial Informatics, (2025)","DOI":"10.1109\/TII.2025.3577686"},{"key":"1893_CR46","doi-asserted-by":"crossref","unstructured":"Chen, H., Huang, Z., Hong, Y., Wang, Y., Lyu, Z., Zhuoer, X., Lan, J., Zhangxuan, G.: Efficient transfer learning for video-language foundation models. In Proceedings of the Computer Vision and Pattern Recognition Conference 29129\u201329138 (2025)","DOI":"10.1109\/CVPR52734.2025.02712"},{"key":"1893_CR47","doi-asserted-by":"crossref","unstructured":"Hou, P., Li, G., Cai, Z., Zhang, J., Huang, D.: Ast-adapter: Parameter-efficient video-to-video transfer learning with adaptive spatiotemporal information bias. IEEE Trans. Circuits Syst. Video Technol. (2025)","DOI":"10.1109\/TCSVT.2025.3638017"},{"key":"1893_CR48","doi-asserted-by":"crossref","unstructured":"Hahm, W.J., Jang, S., Kim, H.T., Lee, D., Kim, K.: Tm-adapter: Temporal merge adapter for efficient global temporal modeling. In Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision 6121\u20136131 (2026)","DOI":"10.1109\/WACV61042.2026.00592"},{"key":"1893_CR49","volume-title":"and Yonggang Lu","author":"X Gao","year":"2025","unstructured":"Gao, X., Chang, Z., Kong, D., Zhou, H.: and Yonggang Lu. Multimodal independent prompt clip for action recognition. IEEE Transactions on Multimedia, Mip-clip (2025)"},{"key":"1893_CR50","doi-asserted-by":"crossref","unstructured":"Goyal, R., Kahou, S.E., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., et al.: The\" something something\" video database for learning and evaluating visual common sense. In Proceedings of the IEEE international conference on computer vision 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"1893_CR51","doi-asserted-by":"crossref","unstructured":"Heilbron, F.C., Escorcia, V., Ghanem, B., Niebles, J.C.: Activitynet: A large-scale video benchmark for human activity understanding. In CVPR 961\u2013970 (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"1893_CR52","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., Serre, T.: Hmdb: a large video database for human motion recognition. In 2011 International conference on computer vision, pages 2556\u20132563. IEEE, (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"issue":"6","key":"1893_CR53","doi-asserted-by":"publisher","first-page":"4653","DOI":"10.1109\/TCSVT.2023.3327605","volume":"34","author":"X Nie","year":"2023","unstructured":"Nie, X., Ni, B., Chang, J., Meng, G., Huo, C., Xiang, S., Tian, Q.: Pro-tuning: Unified prompt tuning for vision tasks. IEEE Trans. Circ. Syst. Video Technol. 34(6), 4653\u20134667 (2023)","journal-title":"IEEE Trans. Circ. Syst. Video Technol."},{"key":"1893_CR54","doi-asserted-by":"crossref","unstructured":"Xia, B., Wu, W., Wang, H., Su, R., He, D., Yang, H., Fan, X., Ouyang, W.: Nsnet: Non-saliency suppression sampler for efficient video recognition. In ECCV, pages 705\u2013723. Springer, (2022)","DOI":"10.1007\/978-3-031-19830-4_40"},{"key":"1893_CR55","doi-asserted-by":"crossref","unstructured":"Xia, B., Wang, Z., Wu, W., Wang, H., Han, J.: Temporal saliency query network for efficient video recognition. In ECCV, pages 741\u2013759. Springer, (2022)","DOI":"10.1007\/978-3-031-19830-4_42"},{"key":"1893_CR56","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, Y., He, Y., Li, Y., Wang, Y., Wang, L., Qiao, Yu.: Uniformerv2: Unlocking the potential of image vits for video understanding. In Proceedings of the IEEE\/CVF International Conference on Computer Vision 1632\u20131643 (2023)","DOI":"10.1109\/ICCV51070.2023.00157"},{"key":"1893_CR57","doi-asserted-by":"crossref","unstructured":"Ruan, X., Yin, Q., Su, F., Zhao, Z.: Eva: Enabling video attributes with hierarchical prompt tuning for action recognition. IEEE Signal Processing Letters, (2025)","DOI":"10.1109\/LSP.2025.3533307"},{"key":"1893_CR58","doi-asserted-by":"crossref","unstructured":"Diba, A., Fayyaz, M., Vivek Sharma, M., Arzani, M., Yousefzadeh, R., Gall, J., Van Gool, L.: Spatio-temporal channel correlation networks for action classification. In Proceedings of the European conference on computer vision (ECCV) 284\u2013299 (2018)","DOI":"10.1007\/978-3-030-01225-0_18"},{"key":"1893_CR59","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., Paluri, M.: A closer look at spatiotemporal convolutions for action recognition. In CVPR 6450\u20136459 (2018)","DOI":"10.1109\/CVPR.2018.00675"},{"key":"1893_CR60","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In proceedings of the IEEE Conference on Computer Vision and Pattern Recognition 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"1893_CR61","first-page":"13098","volume":"34","author":"D Linchao Zhu","year":"2020","unstructured":"Linchao Zhu, D., Tran, L.S.-L., Yang, Y., Feiszli, M., Wang, H.: Faster recurrent networks for efficient video classification. Proc. AAAI Conf. Artificial Intell. 34, 13098\u201313105 (2020)","journal-title":"Proc. AAAI Conf. Artificial Intell."},{"key":"1893_CR62","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Xiong, Y., Liu, W., Lin, D.: Omni-sourced webly-supervised learning for video recognition. In European conference on computer vision 670\u2013688 (2020) Springer,","DOI":"10.1007\/978-3-030-58555-6_40"},{"key":"1893_CR63","doi-asserted-by":"crossref","unstructured":"Zach, C., Pock, T., Bischof, H.: A duality based approach for realtime tv-l 1 optical flow. In Joint pattern recognition symposium 214\u2013223 (2007) Springer,","DOI":"10.1007\/978-3-540-74936-3_22"},{"key":"1893_CR64","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929, (2020)"},{"key":"1893_CR65","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In CVPR 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"1893_CR66","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In Proceedings of the IEEE\/CVF international conference on computer vision 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"}],"container-title":["Journal of Real-Time Image Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-026-01893-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11554-026-01893-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11554-026-01893-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T13:41:33Z","timestamp":1782394893000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11554-026-01893-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,28]]},"references-count":66,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["1893"],"URL":"https:\/\/doi.org\/10.1007\/s11554-026-01893-1","relation":{},"ISSN":["1861-8200","1861-8219"],"issn-type":[{"value":"1861-8200","type":"print"},{"value":"1861-8219","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,28]]},"assertion":[{"value":"17 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"107"}}