{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T04:01:53Z","timestamp":1746331313614,"version":"3.40.4"},"reference-count":50,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"5","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Fundamentals"],"published-print":{"date-parts":[[2025,5,1]]},"DOI":"10.1587\/transfun.2024eap1080","type":"journal-article","created":{"date-parts":[[2024,10,27]],"date-time":"2024-10-27T22:10:40Z","timestamp":1730067040000},"page":"736-745","source":"Crossref","is-referenced-by-count":0,"title":["MST-Adapter: Multi-Scaled Spatio-Temporal Adapter for Parameter-Efficient Image-to-Video Transfer Learning"],"prefix":"10.1587","volume":"E108.A","author":[{"given":"Chenrui","family":"CHANG","sequence":"first","affiliation":[{"name":"Hubei Key Laboratory of Intelligent Robot, the School of Computer Science and Engineering, Wuhan Institute of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tongwei","family":"LU","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Intelligent Robot, the School of Computer Science and Engineering, Wuhan Institute of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"YAO","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Intelligent Robot, the School of Computer Science and Engineering, Wuhan Institute of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] H. Wang and C. Schmid, \u201cAction recognition with improved trajectories,\u201d Proc. IEEE International Conference on Computer Vision, pp.3551-3558, 2013. 10.1109\/iccv.2013.441","DOI":"10.1109\/ICCV.2013.441"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] H. Wang, A. Kl\u00e4ser, C. Schmid, and C.L. Liu, \u201cDense trajectories and motion boundary descriptors for action recognition,\u201d Int. J. Comput. Vis., vol.103, pp.60-79, 2013. 10.1007\/s11263-012-0594-8","DOI":"10.1007\/s11263-012-0594-8"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] C. Feichtenhofer, A. Pinz, and A. Zisserman, \u201cConvolutional two-stream network fusion for video action recognition,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.1933-1941, 2016. 10.1109\/cvpr.2016.213","DOI":"10.1109\/CVPR.2016.213"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] C. Feichtenhofer, A. Pinz, and R.P. Wildes, \u201cSpatiotemporal multiplier networks for video action recognition,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.4768-4777, 2017. 10.1109\/cvpr.2017.787","DOI":"10.1109\/CVPR.2017.787"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] L. Wang, Y. Xiong, Z. Wang, Y. Qiao, D. Lin, X. Tang, and L. Van Gool, \u201cTemporal segment networks: Towards good practices for deep action recognition,\u201d European Conference on Computer Vision, pp.20-36, Springer, 2016. 10.1007\/978-3-319-46484-8_2","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] Y. Wang, M. Long, J. Wang, and P.S. Yu, \u201cSpatiotemporal pyramid network for video action recognition,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.1529-1538, 2017. 10.1109\/cvpr.2017.226","DOI":"10.1109\/CVPR.2017.226"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] D. Neimark, O. Bar, M. Zohar, and D. Asselmann, \u201cVideo transformer network,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.3163-3172, 2021. 10.1109\/iccvw54120.2021.00355","DOI":"10.1109\/ICCVW54120.2021.00355"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] R. Girdhar, J. Carreira, C. Doersch, and A. Zisserman, \u201cVideo action transformer network,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.244-253, 2019. 10.1109\/cvpr.2019.00033","DOI":"10.1109\/CVPR.2019.00033"},{"key":"9","unstructured":"[9] G. Bertasius, H. Wang, and L. Torresani, \u201cIs space-time attention all you need for video understanding?,\u201d ICML, p.4, 2021."},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] H. Fan, B. Xiong, K. Mangalam, Y. Li, Z. Yan, J. Malik, and C. Feichtenhofer, \u201cMultiscale vision transformers,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.6824-6835, 2021. 10.1109\/iccv48922.2021.00675","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] Y. Li, C.Y. Wu, H. Fan, K. Mangalam, B. Xiong, J. Malik, and C. Feichtenhofer, \u201cMviTv2: Improved multiscale vision transformers for classification and detection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.4804-4814, 2022. 10.1109\/cvpr52688.2022.00476","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] Z. Liu, J. Ning, Y. Cao, Y. Wei, Z. Zhang, S. Lin, and H. Hu, \u201cVideo swin transformer,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.3202-3211, 2022. 10.1109\/cvpr52688.2022.00320","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] C. Feichtenhofer, H. Fan, J. Malik, and K. He, \u201cSlowFast networks for video recognition,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.6202-6211, 2019. 10.1109\/iccv.2019.00630","DOI":"10.1109\/ICCV.2019.00630"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] S. Yan, X. Xiong, A. Arnab, Z. Lu, M. Zhang, C. Sun, and C. Schmid, \u201cMultiview transformers for video recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.3333-3343, 2022. 10.1109\/cvpr52688.2022.00333","DOI":"10.1109\/CVPR52688.2022.00333"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] C. Yang, Y. Xu, J. Shi, B. Dai, and B. Zhou, \u201cTemporal pyramid network for action recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.591-600, 2020. 10.1109\/cvpr42600.2020.00067","DOI":"10.1109\/CVPR42600.2020.00067"},{"key":"16","unstructured":"[16] A. Krizhevsky, I. Sutskever, and G.E. Hinton, \u201cImagenet classification with deep convolutional neural networks,\u201d Advances in Neural Information Processing Systems, vol.25, 2012."},{"key":"17","unstructured":"[17] W. Kay, J. Carreira, K. Simonyan, B. Zhang, C. Hillier, S. Vijayanarasimhan, F. Viola, T. Green, T. Back, P. Natsev, M. Suleyman, and A. Zisserman, \u201cThe Kinetics human action video dataset,\u201d arXiv preprint arXiv:1705.06950, 2017. 10.48550\/arXiv.1705.06950"},{"key":"18","unstructured":"[18] N. Houlsby, A. Giurgiu, S. Jastrzebski, B. Morrone, Q. De Laroussilhe, A. Gesmundo, M. Attariyan, and S. Gelly, \u201cParameter-efficient transfer learning for NLP,\u201d International Conference on Machine Learning, pp.2790-2799, PMLR, 2019."},{"key":"19","unstructured":"[19] X.L. Li and P. Liang, \u201cPrefix-tuning: Optimizing continuous prompts for generation,\u201d arXiv preprint arXiv:2101.00190, 2021. 10.48550\/arXiv.2101.00190"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] B. Lester, R. Al-Rfou, and N. Constant, \u201cThe power of scale for parameter-efficient prompt tuning,\u201d arXiv preprint arXiv:2104.08691, 2021. 10.48550\/arXiv.2104.08691","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"21","unstructured":"[21] E.J. Hu, Y. Shen, P. Wallis, Z. Allen-Zhu, Y. Li, S. Wang, L. Wang, and W. Chen, \u201cLoRA: Low-rank adaptation of large language models,\u201d arXiv preprint arXiv:2106.09685, 2021. 10.48550\/arXiv.2106.09685"},{"key":"22","unstructured":"[22] S. Chen, C. Ge, Z. Tong, J. Wang, Y. Song, J. Wang, and P. Luo, \u201cAdaptformer: Adapting vision transformers for scalable visual recognition,\u201d Advances in Neural Information Processing Systems, vol.35, pp.16664-16678, 2022."},{"key":"23","unstructured":"[23] X. He, C. Li, P. Zhang, J. Yang, and X.E. Wang, \u201cParameter-efficient model adaptation for vision transformers,\u201d arXiv preprint arXiv:2203.16329, 2022. 10.48550\/arXiv.2203.16329"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] M. Jia, L. Tang, B.C. Chen, C. Cardie, S. Belongie, B. Hariharan, and S.N. Lim, \u201cVisual prompt tuning,\u201d European Conference on Computer Vision, pp.709-727, Springer, 2022. 10.1007\/978-3-031-19827-4_41","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"25","unstructured":"[25] J. Pan, Z. Lin, X. Zhu, J. Shao, and H. Li, \u201cST-adapter: Parameter-efficient image-to-video transfer learning,\u201d Advances in Neural Information Processing Systems, vol.35, pp.26462-26477, 2022."},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] Z. Qing, S. Zhang, Z. Huang, Y. Zhang, C. Gao, D. Zhao, and N. Sang, \u201cDisentangling spatial and temporal learning for efficient image-to-video transfer learning,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.13934-13944, 2023. 10.1109\/iccv51070.2023.01281","DOI":"10.1109\/ICCV51070.2023.01281"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] Z. Lin, S. Geng, R. Zhang, P. Gao, G. de Melo, X. Wang, J. Dai, Y. Qiao, and H. Li, \u201cFrozen clip models are efficient video learners,\u201d European Conference on Computer Vision, pp.388-404, Springer, 2022. 10.1007\/978-3-031-19833-5_23","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"28","unstructured":"[28] T. Yang, Y. Zhu, Y. Xie, A. Zhang, C. Chen, and M. Li, \u201cAIM: Adapting image models for efficient video action recognition,\u201d arXiv preprint arXiv:2302.03024, 2023. 10.48550\/arXiv.2302.03024"},{"key":"29","unstructured":"[29] A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, and N. Houlsby, \u201cAn image is worth 16x16 words: Transformers for image recognition at scale,\u201d arXiv preprint arXiv:2010.11929, 2020. 10.48550\/arXiv.2010.11929"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, and B. Guo, \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.10012-10022, 2021. 10.1109\/iccv48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] Z. Liu, H. Hu, Y. Lin, Z. Yao, Z. Xie, Y. Wei, J. Ning, Y. Cao, Z. Zhang, L. Dong, F. Wei, and B. Guo, \u201cSwin transformer V2: Scaling up capacity and resolution,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.12009-12019, 2022. 10.1109\/cvpr52688.2022.01170","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"32","doi-asserted-by":"crossref","unstructured":"[32] J. Deng, W. Dong, R. Socher, L.J. Li, K. Li, and L. Fei-Fei, \u201cImageNet: A large-scale hierarchical image database,\u201d 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp.248-255, IEEE, 2009. 10.1109\/cvpr.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"33","unstructured":"[34] K. Simonyan and A. Zisserman, \u201cTwo-stream convolutional networks for action recognition in videos,\u201d Advances in Neural Information Processing Systems, vol.27, 2014."},{"key":"34","doi-asserted-by":"crossref","unstructured":"[35] D. Tran, L. Bourdev, R. Fergus, L. Torresani, and M. Paluri, \u201cLearning spatiotemporal features with 3d convolutional networks,\u201d Proc. IEEE International Conference on Computer Vision, pp.4489-4497, 2015. 10.1109\/iccv.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"35","doi-asserted-by":"crossref","unstructured":"[36] Y. Liang, P. Zhou, R. Zimmermann, and S. Yan, \u201cDualFormer: Local-global stratified transformer for efficient video recognition,\u201d European Conference on Computer Vision, pp.577-595, Springer, 2022. 10.1007\/978-3-031-19830-4_33","DOI":"10.1007\/978-3-031-19830-4_33"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[37] W. Xiang, C. Li, B. Wang, X. Wei, X.S. Hua, and L. Zhang, \u201cSpatiotemporal self-attention modeling with temporal patch shift for action recognition,\u201d European Conference on Computer Vision, pp.627-644, Springer, 2022. 10.1007\/978-3-031-20062-5_36","DOI":"10.1007\/978-3-031-20062-5_36"},{"key":"37","doi-asserted-by":"crossref","unstructured":"[38] C. Wei, H. Fan, S. Xie, C.Y. Wu, A. Yuille, and C. Feichtenhofer, \u201cMasked feature prediction for self-supervised visual pre-training,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.14668-14678, 2022. 10.1109\/cvpr52688.2022.01426","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"38","doi-asserted-by":"crossref","unstructured":"[39] Y.L. Sung, J. Cho, and M. Bansal, \u201cVL-ADAPTER: Parameter-efficient transfer learning for vision-and-language tasks,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.5227-5237, 2022. 10.1109\/cvpr52688.2022.00516","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"39","unstructured":"[40] J.L. Ba, J.R. Kiros, and G.E. Hinton, \u201cLayer normalization,\u201d arXiv preprint arXiv:1607.06450, 2016. 10.48550\/arXiv.1607.06450"},{"key":"40","unstructured":"[41] D. Hendrycks and K. Gimpel, \u201cGaussian error linear units (GELUs),\u201d arXiv preprint arXiv:1606.08415, 2016. 10.48550\/arXiv.1606.08415"},{"key":"41","unstructured":"[42] A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, and I. Sutskever, \u201cLearning transferable visual models from natural language supervision,\u201d International Conference on Machine Learning, pp.8748-8763, PMLR, 2021."},{"key":"42","doi-asserted-by":"crossref","unstructured":"[43] C. Szegedy, W. Liu, Y. Jia, P. Sermanet, S. Reed, D. Anguelov, D. Erhan, V. Vanhoucke, and A. Rabinovich, \u201cGoing deeper with convolutions,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.1-9, 2015. 10.1109\/cvpr.2015.7298594","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"43","doi-asserted-by":"crossref","unstructured":"[44] J. Lin, C. Gan, and S. Han, \u201cTSM: Temporal shift module for efficient video understanding,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.7083-7093, 2019. 10.1109\/iccv.2019.00718","DOI":"10.1109\/ICCV.2019.00718"},{"key":"44","doi-asserted-by":"crossref","unstructured":"[45] H. Zhang, Y. Hao, and C.W. Ngo, \u201cToken shift transformer for video classification,\u201d Proc. 29th ACM International Conference on Multimedia, pp.917-925, 2021. 10.1145\/3474085.3475272","DOI":"10.1145\/3474085.3475272"},{"key":"45","doi-asserted-by":"crossref","unstructured":"[46] R. Goyal, S. Ebrahimi Kahou, V. Michalski, J. Materzynska, S. Westphal, H. Kim, V. Haenel, I. Fruend, P. Yianilos, M. Mueller-Freitag, F. Hoppe, C. Thurau, I. Bax, and R. Memisevic, \u201cThe \u201csomething something\u201d video database for learning and evaluating visual common sense,\u201d Proc. IEEE International Conference on Computer Vision, pp.5842-5850, 2017. 10.1109\/iccv.2017.622","DOI":"10.1109\/ICCV.2017.622"},{"key":"46","unstructured":"[47] M. Contributors, \u201cOpenmmlab\u2019s next generation video understanding toolbox and benchmark,\u201d https:\/\/github.com\/open-mmlab\/mmaction2, 2020."},{"key":"47","unstructured":"[48] I. Loshchilov and F. Hutter, \u201cDecoupled weight decay regularization,\u201d arXiv preprint arXiv:1711.05101, 2017. 10.48550\/arXiv.1711.05101"},{"key":"48","unstructured":"[49] I. Loshchilov and F. Hutter, \u201cSGDR: Stochastic gradient descent with warm restarts,\u201d arXiv preprint arXiv:1608.03983, 2016. 10.48550\/arXiv.1608.03983"},{"key":"49","doi-asserted-by":"crossref","unstructured":"[50] A. Arnab, M. Dehghani, G. Heigold, C. Sun, M. Lu\u010di\u0107, and C. Schmid, \u201cViViT: A video vision transformer,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.6836-6846, 2021. 10.1109\/iccv48922.2021.00676","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"50","unstructured":"[51] K. Li, Y. Wang, P. Gao, G. Song, Y. Liu, H. Li, and Y. Qiao, \u201cUniFormer: Unified transformer for efficient spatiotemporal representation learning,\u201d arXiv preprint arXiv:2201.04676, 2022. 10.48550\/arXiv.2201.04676"}],"container-title":["IEICE Transactions on Fundamentals of Electronics, Communications and Computer Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transfun\/E108.A\/5\/E108.A_2024EAP1080\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,3]],"date-time":"2025-05-03T03:30:55Z","timestamp":1746243055000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transfun\/E108.A\/5\/E108.A_2024EAP1080\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,1]]},"references-count":50,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transfun.2024eap1080","relation":{},"ISSN":["0916-8508","1745-1337"],"issn-type":[{"type":"print","value":"0916-8508"},{"type":"electronic","value":"1745-1337"}],"subject":[],"published":{"date-parts":[[2025,5,1]]},"article-number":"2024EAP1080"}}