{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T17:06:15Z","timestamp":1772643975789,"version":"3.50.1"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T00:00:00Z","timestamp":1764979200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T00:00:00Z","timestamp":1764979200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["No. 2020YFC1522900"],"award-info":[{"award-number":["No. 2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["No. cstc2021jscxgksbX0056"],"award-info":[{"award-number":["No. cstc2021jscxgksbX0056"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s00371-025-04287-9","type":"journal-article","created":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T04:34:49Z","timestamp":1764995689000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TD4V: temporal difference module for efficient video action recognition via fine-tuning and side-tuning"],"prefix":"10.1007","volume":"42","author":[{"given":"Youwei","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junyong","family":"Ye","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyi","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingjing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"4287_CR1","doi-asserted-by":"crossref","unstructured":"J. Huang, L. Teng, Y. Xiao, A. Zhu, X. Liu, Lip reading using temporal adaptive module, in: neural information processing - 30th international conference, ICONIP, pp. 347\u2013356.","DOI":"10.1007\/978-981-99-8141-0_26"},{"key":"4287_CR2","doi-asserted-by":"publisher","first-page":"6821","DOI":"10.1109\/TMM.2022.3214776","volume":"25","author":"J Zhu","year":"2022","unstructured":"Zhu, J., Zhang, Q., Fei, L., Cai, R., Xie, Y., Sheng, B., Yang, X.: FFFN: frame-by-frame feedback fusion network for video super-resolution. IEEE Trans. Multimedia 25, 6821\u20136835 (2022)","journal-title":"IEEE Trans. Multimedia"},{"issue":"10","key":"4287_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3679203","volume":"20","author":"S Chen","year":"2024","unstructured":"Chen, S., Zhong, X., Zhang, Y., Zhu, L., Li, P., Yang, X., Sheng, B.: Action-aware linguistic skeleton optimization network for non-autoregressive video captioning. ACM Trans. Multim. Comput. Commun. Appl. 20(10), 1\u201324 (2024). https:\/\/doi.org\/10.1145\/3679203","journal-title":"ACM Trans. Multim. Comput. Commun. Appl."},{"key":"4287_CR4","doi-asserted-by":"publisher","unstructured":"H. Wang, C. Schmid, Action recognition with improved trajectories, In: IEEE International Conference on Computer Vision, ICCV 2013, pp. 3551\u20133558, https:\/\/doi.org\/10.1109\/ICCV.2013.441.","DOI":"10.1109\/ICCV.2013.441"},{"key":"4287_CR5","doi-asserted-by":"publisher","unstructured":"K. Simonyan, A. Zisserman, Two-stream convolutional networks for action recognition in videos, In: Advances in Neural Information Processing Systems 27, NeurIPS 2014, pp. 568\u2013576, https:\/\/doi.org\/10.48550\/arXiv.1406.2199.","DOI":"10.48550\/arXiv.1406.2199"},{"key":"4287_CR6","doi-asserted-by":"publisher","unstructured":"D. Tran, H. Wang, L. Torresani, J. Ray, Y. LeCun, M. Paluri, A closer look at spatiotemporal convolutions for action recognition, In: 2018 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2018, pp. 6450\u20136459, https:\/\/doi.org\/10.1109\/CVPR.2018.00675.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"4287_CR7","doi-asserted-by":"publisher","unstructured":"H. Tani, Graph convolutional networks with minimal appearance information for action recognition, In: IEEE International Conference on Image Processing, ICIP 2024, pp. 388\u2013394, https:\/\/doi.org\/10.1109\/ICIP51287.2024.10647978.","DOI":"10.1109\/ICIP51287.2024.10647978"},{"key":"4287_CR8","doi-asserted-by":"publisher","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, N. Houlsby, An image is worth 16x16 words: Transformers for image recognition at scale, In: 9th International Conference on Learning Representations, ICLR 2021, pp. 1\u201321, https:\/\/doi.org\/10.48550\/arXiv.2010.11929.","DOI":"10.48550\/arXiv.2010.11929"},{"key":"4287_CR9","unstructured":"A. Radford, J. W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, I. Sutskever, Learning transferable visual models from natural language supervision, In: Proceedings of the 38th International Conference on Machine Learning, ICML 2021, pp. 8748\u20138763, URL: http:\/\/proceedings.mlr.press\/v139\/radford21a.html."},{"key":"4287_CR10","doi-asserted-by":"publisher","first-page":"5410","DOI":"10.1109\/TMM.2023.3333206","volume":"26","author":"Y Hu","year":"2024","unstructured":"Hu, Y., Gao, J., Dong, J., Fan, B., Liu, H.: Exploring rich semantics for open-set action recognition. IEEE Trans. Multim. 26, 5410\u20135421 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3333206","journal-title":"IEEE Trans. Multim."},{"key":"4287_CR11","doi-asserted-by":"publisher","first-page":"555","DOI":"10.1109\/TMM.2023.3267887","volume":"26","author":"Y Hu","year":"2024","unstructured":"Hu, Y., Gao, J., Xu, C.: Learning multi-expert distribution calibration for long-tailed video classification. IEEE Trans. Multim. 26, 555\u2013567 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3267887","journal-title":"IEEE Trans. Multim."},{"key":"4287_CR12","first-page":"12786","volume":"34","author":"M Ryoo","year":"2021","unstructured":"Ryoo, M., Piergiovanni, A.J., Arnab, A., Dehghani, M., Angelova, A.: TokenLearner: Adaptive space-time tokenization for videos. Adv. Neural Inf. Process. Syst. 34, 12786\u201312797 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4287_CR13","doi-asserted-by":"publisher","first-page":"16664","DOI":"10.52202\/068431-1212","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen, S., Ge, C., Tong, Z., Wang, J., Song, Y., Wang, J., Luo, P.: AdaptFormer: Adapting vision transformers for scalable visual recognition. Adv. Neural Inf. Process. Syst. 35, 16664\u201316678 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4287_CR14","doi-asserted-by":"publisher","unstructured":"M. Jia, L. Tang, B. Chen, C. Cardie, S. J. Belongie, B. Hariharan, S. Lim, Visual prompt tuning, In: Proceedings of European Conference on Computer Vision, ECCV 2022, pp. 709\u2013727, https:\/\/doi.org\/10.1007\/978-3-031-19827-4_41.","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"4287_CR15","doi-asserted-by":"publisher","unstructured":"Z. Qing, S. Zhang, Z. Huang, Y. Zhang, C. Gao, D. Zhao, N. Sang, Disentangling spatial and temporal learning for efficient image-to-video transfer learning, In: IEEE\/CVF International Conference on Computer Vision, ICCV 2023, pp. 13888\u201313895, https:\/\/doi.org\/10.1109\/ICCV51070.2023.01281.","DOI":"10.1109\/ICCV51070.2023.01281"},{"key":"4287_CR16","doi-asserted-by":"publisher","unstructured":"H. Yao, W. Wu, Z. Li, Side4Video: Spatial-temporal side network for memory-efficient image-to-video transfer learning, arxiv preprint 2023, pp. 1\u201314, https:\/\/doi.org\/10.48550\/arXiv.2311.15769.","DOI":"10.48550\/arXiv.2311.15769"},{"key":"4287_CR17","doi-asserted-by":"publisher","unstructured":"R. Goyal, S. E. Kahou, V. Michalski, J. Materzynska, S. Westphal, H. Kim, V. Haenel, I. Fr\u00fcnd, P. Yianilos, M. Mueller-Freitag, F. Hoppe, C. Thurau, I. Bax, R. Memisevic, The \"something something\" video database for learning and evaluating visual common sense, in: IEEE International Conference on Computer Vision, ICCV 2017, pp. 5843\u20135851, https:\/\/doi.org\/10.1109\/ICCV.2017.622.","DOI":"10.1109\/ICCV.2017.622"},{"key":"4287_CR18","doi-asserted-by":"publisher","unstructured":"H. Kuehne, H. Jhuang, E. Garrote, T. A. Poggio, T. Serre, HMDB: A large video database for human motion recognition, in: IEEE International Conference on Computer Vision, ICCV 2011, pp. 2556\u20132563, https:\/\/doi.org\/10.1109\/ICCV.2011.6126543.","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"4287_CR19","doi-asserted-by":"publisher","unstructured":"K. Soomro, A. R. Zamir, M. Shah, UCF101: A dataset of 101 human actions classes from videos in the wild, arxiv preprint 2012, pp. 1\u20137, https:\/\/doi.org\/10.48550\/arXiv.1212.0402.","DOI":"10.48550\/arXiv.1212.0402"},{"key":"4287_CR20","doi-asserted-by":"crossref","unstructured":"Y. Li, Y. Li, N. Vasconcelos, RESOUND: Towards action recognition without representation bias, In: Proceedings of European Conference on Computer Vision, ECCV 2018, pp. 520\u2013535.","DOI":"10.1007\/978-3-030-01231-1_32"},{"key":"4287_CR21","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. Kaiser, I. Polosukhin, Attention is all you need, In: Neural information processing systems 30, NeurIPS 2017, pp. 5998\u20136008, URL: https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html."},{"key":"4287_CR22","unstructured":"G. Bertasius, H. Wang, L. Torresani, Is space-time attention all you need for video understanding?, In: Proceedings of the 38th International Conference on Machine Learning, ICML 2021, pp. 813\u2013824, URL: http:\/\/proceedings.mlr.press\/v139\/bertasius21a.html."},{"key":"4287_CR23","doi-asserted-by":"publisher","unstructured":"A. Arnab, M. Dehghani, G. Heigold, C. Sun, M. Lucic, C. Schmid, ViViT: A video vision transformer, In: 2021 IEEE\/CVF International Conference on Computer Vision, ICCV 2021, pp. 6816\u20136826, https:\/\/doi.org\/10.1109\/ICCV48922.2021.00676.","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"4287_CR24","doi-asserted-by":"publisher","unstructured":"Y. Zhang, X. Li, C. Liu, B. Shuai, Y. Zhu, B. Brattoli, H. Chen, I. Marsic, J. Tighe, VidTr: Video transformer without convolutions, In: 2021 IEEE\/CVF International Conference on Computer Vision, ICCV 2021, pp. 13557\u201313567, https:\/\/doi.org\/10.1109\/ICCV48922.2021.01332.","DOI":"10.1109\/ICCV48922.2021.01332"},{"issue":"9","key":"4287_CR25","doi-asserted-by":"publisher","first-page":"6279","DOI":"10.1007\/S00371-023-03165-6","volume":"40","author":"W Li","year":"2024","unstructured":"Li, W., Gong, W., Qian, Y., Tian, H.: STAM: a spatio-temporal adaptive module for improving static convolutions in action recognition. Vis. Comput. 40(9), 6279\u20136293 (2024). https:\/\/doi.org\/10.1007\/S00371-023-03165-6","journal-title":"Vis. Comput."},{"key":"4287_CR26","doi-asserted-by":"publisher","unstructured":"M. Kim, P. H. Seo, C. Schmid, M. Cho, Learning correlation structures for vision transformers, In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, pp. 18941\u201318951, https:\/\/doi.org\/10.48550\/arXiv.2404.03924.","DOI":"10.48550\/arXiv.2404.03924"},{"key":"4287_CR27","unstructured":"N. Houlsby, A. Giurgiu, S. Jastrzebski, B. Morrone, Q. Laroussilhe, A. Gesmundo, M. Attariyan, S. Gelly, Parameter-efficient transfer learning for NLP, In: Proceedings of the 36th International Conference on Machine Learning, ICML 2019, pp. 2790\u20132799, URL: http:\/\/proceedings.mlr.press\/v97\/houlsby19a.html."},{"key":"4287_CR28","first-page":"26462","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., Li, H.: ST-Adapter: Parameter-efficient image-to-video transfer learning. Adv. Neural Inf. Process. Syst. 35, 26462\u201326477 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4287_CR29","unstructured":"T. Yang, Y. Zhu, Y. Xie, A. Zhang, C. Chen, M. Li, AIM: Adapting image models for efficient video action recognition, In: The Eleventh International Conference on Learning Representations, ICLR 2023, pp. 1\u201318."},{"key":"4287_CR30","doi-asserted-by":"publisher","unstructured":"J. Park, J. Lee, K. Sohn, Dual-Path Adaptation from image to video transformers, In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, pp. 2203\u20132213, https:\/\/doi.org\/10.1109\/CVPR52729.2023.00219.","DOI":"10.1109\/CVPR52729.2023.00219"},{"key":"4287_CR31","doi-asserted-by":"crossref","unstructured":"C. Ju, T. Han, K. Zheng, Y. Zhang, W. Xie, Prompting visual-language models for efficient video understanding, In: Proceedings of European Conference on Computer Vision, ECCV 2022, pp. 105\u2013124.","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"4287_CR32","doi-asserted-by":"publisher","unstructured":"S. T. Wasim, M. Naseer, S. H. Khan, F. S. Khan, M. Shah, Vita-CLIP: video and text adaptive clip via multimodal prompting, In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, pp. 23034\u201323044, https:\/\/doi.org\/10.1109\/CVPR52729.2023.02206.","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"4287_CR33","first-page":"12991","volume":"35","author":"YL Sung","year":"2022","unstructured":"Sung, Y.L., Cho, J., Bansal, M.: LST: Ladder side-tuning for parameter and memory efficient transfer learning. Adv. Neural Inf. Process. Syst. 35, 12991\u201313005 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"4287_CR34","doi-asserted-by":"publisher","unstructured":"R. Liu, J. Huang, G. Li, J. Feng, X. Wu, T. H. Li, Revisiting temporal modeling for clip-based image-to-video knowledge transferring, In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, pp. 6555\u20136564, https:\/\/doi.org\/10.1109\/CVPR52729.2023.00634.","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"4287_CR35","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-024-03478-0","author":"C Liu","year":"2024","unstructured":"Liu, C., Gu, F.: Differential motion attention network for efficient action recognition. Vis. Comput. (2024). https:\/\/doi.org\/10.1007\/s00371-024-03478-0","journal-title":"Vis. Comput."},{"key":"4287_CR36","doi-asserted-by":"publisher","unstructured":"E. B. Zaken, Y. Goldberg, S. Ravfogel, BitFit: Simple parameter-efficient fine-tuning for transformer-based masked language-models, In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), ACL 2022, pp. 1\u20139, https:\/\/doi.org\/10.18653\/V1\/2022.ACL-SHORT.1.","DOI":"10.18653\/V1\/2022.ACL-SHORT.1"},{"key":"4287_CR37","doi-asserted-by":"publisher","unstructured":"W. Wu, X. Wang, H. Luo, J. Wang, Y. Yang, W. Ouyang, Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models, In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, pp. 6620\u20136630, https:\/\/doi.org\/10.1109\/CVPR52729.2023.00640.","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"4287_CR38","doi-asserted-by":"publisher","unstructured":"M. Sandler, A. G. Howard, M. Zhu, A. Zhmoginov, L. Chen, MobileNetV2: Inverted residuals and linear bottlenecks, In: 2018 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2018, pp. 4510\u20134520, https:\/\/doi.org\/10.1109\/CVPR.2018.00474.","DOI":"10.1109\/CVPR.2018.00474"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04287-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04287-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04287-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T13:01:56Z","timestamp":1772629316000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04287-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":38,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["4287"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04287-9","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"19 January 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"15"}}