{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T05:29:21Z","timestamp":1776922161706,"version":"3.51.2"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,5,16]],"date-time":"2025-05-16T00:00:00Z","timestamp":1747353600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,16]],"date-time":"2025-05-16T00:00:00Z","timestamp":1747353600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFC1522900"],"award-info":[{"award-number":["2020YFC1522900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]},{"name":"Chongqing Technological Innovation and Application Development Project","award":["cstc2021jscx-gksbX0056"],"award-info":[{"award-number":["cstc2021jscx-gksbX0056"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00530-025-01836-z","type":"journal-article","created":{"date-parts":[[2025,5,16]],"date-time":"2025-05-16T16:36:23Z","timestamp":1747413383000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["MLKD-CLIP: Multi-layer Feature Knowledge Distillation of CLIP for Open-vocabulary Action Recognition"],"prefix":"10.1007","volume":"31","author":[{"given":"Jingjing","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junyong","family":"Ye","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youwei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyi","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaoming","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,16]]},"reference":[{"key":"1836_CR1","unstructured":"Jianzong, W., Xiangtai, L., Shilin, X.*, Haobo, Y.*, Henghui, D., Yibo, Y., Xia, L., Jiangning, Z., Yunhai, T., Xudong, J., Bernard, G., Dacheng, T.: Towards open vocabulary learning: A Survey. In IEEE transactions on pattern analysis and machine intelligence (2024)."},{"key":"1836_CR2","unstructured":"Wentao, B, Qi, Y., Yu, K.: OpenTAL: Towards open set temporal action localization. In conference on computer vision and pattern recognition (2022)."},{"key":"1836_CR3","unstructured":"Alec, R., Jong, W. K., Chris, H., Aditya, R, Gabriel, Gh, Sandhini, A., Girish, S., Amanda, A., Pamela, M., Jack, C., et al. Learning transferable visual models from natural language supervision. In international conference on machine learning (2021)"},{"key":"1836_CR4","unstructured":"Chao, J., Yinfei, Y., Ye, X., Yi-Ting, C., Zarana, P., Hieu, P., Quoc, L., Yun-Hsuan, S., Zhen, L., and Tom, D.: Scaling up visual and vision-language representation learning with noisy text supervision. In international conference on machine learning (2021)."},{"key":"1836_CR5","unstructured":"Zhaoqing, W., Yu, L., Qiang, L., Xunqiang, T., Yandong, G., Mingming, G., Tongliang, L.: CRIS: CLIP-driven referring image segmentation. In: IEEE Conference on computer vision and pattern recognition (2022)"},{"key":"1836_CR6","unstructured":"Golnaz, G., Xiuye, G., Yin, C., Tsung-Y. L.: Scaling open-vocabulary image segmentation with image-level labels. In: European conference on computer vision (2022)"},{"key":"1836_CR7","unstructured":"Wanfeng, Z., Qiang, L., Xiaoyan, G., Pengfei, W., Zhongyuan, W.: Bridging CLIP and StyleGAN through latent alignment for image editing. arXiv preprint arXiv:2210.04506 (2022)"},{"key":"1836_CR8","unstructured":"Katherine, C., Stella, B., Daniel, K., Dashiell, S., Eric H., Louis, C., Edward, R.: VQGAN-CLIP: open domain image generation and editing with natural language guidance. In: European Conference on Computer Vision (2022)."},{"key":"1836_CR9","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1007\/s00530-024-01414-9","volume":"30","author":"Hairui Yang","year":"2024","unstructured":"Yang, Hairui, Wang, Ning, Li, Haojie, Wang, Lei, Wang, Zhihui: Application of CLIP for efficient zero-shot learning. Multimed. Syst. 30, 219 (2024). https:\/\/doi.org\/10.1007\/s00530-024-01414-9","journal-title":"Multimed. Syst."},{"key":"1836_CR10","unstructured":"Wenxuan, W., Quan, S., Fan, Z., Yepeng, T., Jing, L., Xinlong, W.: Diffusion feedback helps CLIP see better. arXiv preprint arXiv:2407.20171(2024)."},{"key":"1836_CR11","unstructured":"Ashish, V., Noam, S., Niki, P., Jakob, U., Llion, J., Aidan N,. G., Lukasz, K., Illia, P.: Attention is all you need. In conference and workshop on neural information processing systems (2017)"},{"key":"1836_CR12","doi-asserted-by":"publisher","first-page":"12841","DOI":"10.1007\/s00521-020-04730-z","volume":"32","author":"Li Cheng","year":"2020","unstructured":"Cheng, Li., Jing, Xiao-Yuan., Zhu, Xiaoke, Ma, Fei, Chang-Hui, Hu., Cai, Ziyun, Qi, Fumin, et al.: Scale-fusion framework for improving video-based person re-identification performance. Neural. Comput. App. 32, 12841\u201312858 (2020). https:\/\/doi.org\/10.1007\/s00521-020-04730-z","journal-title":"Neural. Comput. App."},{"key":"1836_CR13","unstructured":"Junting, P., Ziyi, L., Xiatian, Z., Jing, S., Hongsheng, L.: ST-Adapter: parameter-efficient image-to-video transfer learning. In conference and workshop on neural information processing systems (2022)"},{"key":"1836_CR14","unstructured":"Taojiannan, Y., Yi, Z., Yusheng, X., Aston, Z., Chen, C., Mu, L.: AIM: adapting image models for efficient video action recognition. In international conference on learning representations (2023)"},{"key":"1836_CR15","unstructured":"Bolin, N., Houwen, P., Minghao, C., Songyang, Z., Gaofeng, M., Jianlong, F., Shiming, X., Haibin, L. Expanding language-image pretrained models for general video recognition. In European conference on computer vision (2022)"},{"key":"1836_CR16","unstructured":"Xiaohu H., Hao Z., Kun, Y, and Kai, H.: Froster: Frozen clip is a strong teacher for open-vocabulary action recognition. In international conference on learning representations (2024)."},{"key":"1836_CR17","unstructured":"Khurram, S., Amir, R. Z., Mubarak, S.: UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402(2012)."},{"key":"1836_CR18","unstructured":"Hildegard, K., Hueihan, J., Est\u00b4\u0131baliz, G., Tomaso, P., and Thomas, S.: Hmdb: a large video database for human motion recognition. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (2011)."},{"key":"1836_CR19","unstructured":"Raghav, G., Samira, E. K., Vincent, M., Joanna, M., Susanne, W., Heuna, K., Valentin, H., Ingo, F., Peter, Y., Moritz, M-F., et al. The\u201d something something\u201d video database for learning and evaluating visual common sense. In Proceedings of the IEEE\/CVF international conference on computer vision (2017)"},{"key":"1836_CR20","unstructured":"Limin, W., Yuanjun, X., Zhe, W., Yu, Q., Dahua, L., Xiaoou, T., Luc Van, G.: Temporal segment networks: towards good practices for deep action recognition. In European conference on computer vision (2016)"},{"key":"1836_CR21","unstructured":"Zhaoyang, L., Donghao, L., Yabiao, W., Limin, W., Ying, T., Chengjie, W., Jilin, L., Feiyue, H., and Tong, L.: TEINet: towards an efficient architecture for video recognition. In association for the advancement of artificial intelligence (2020)"},{"key":"1836_CR22","unstructured":"Ji, L., Chuang, G., Song, H. TSM: Temporal shift module for efficient video understanding. In international conference on computer vision (2019)."},{"key":"1836_CR23","doi-asserted-by":"publisher","first-page":"487","DOI":"10.1007\/s00530-022-00961-3","volume":"29","author":"Aihua Zhou","year":"2023","unstructured":"Zhou, Aihua, Ma, Yujun, Ji, Wanting, Zong, Ming, Yang, Pei, Min, Wu., Liu, Mingzhe: Multi-head attention-based two-stream EfficientNet for action recognition. Multimed. Syst. 29, 487\u2013498 (2023). https:\/\/doi.org\/10.1007\/s00530-022-00961-3","journal-title":"Multimed. Syst."},{"key":"1836_CR24","unstructured":"Joao, C., Andrew, Z., Quo, V.: Action recognition? A new model and the kinetics dataset. In conference on computer vision and pattern recognition (2018)."},{"key":"1836_CR25","unstructured":"Christoph, F., Haoqi, F., Jitendra, M., Kaiming, H.: SlowFast networks for video recognition. In conference on computer vision and pattern recognition (2019)"},{"key":"1836_CR26","unstructured":"Christoph, F.: X3D: Expanding architectures for efficient video recognition. In conference on computer vision and pattern recognition (2020)"},{"issue":"5","key":"1836_CR27","doi-asserted-by":"publisher","first-page":"3067","DOI":"10.32604\/cmc.2024.049512","volume":"79","author":"Arnab Dey","year":"2024","unstructured":"Dey, Arnab, Biswas, Samit, Le, Dac-Nhuong.: Workout action recognition in video streams using an attention driven residual DC-GRU Network. Comput. Mater. Continua. 79(5), 3067\u20133087 (2024)","journal-title":"Comput. Mater. Continua."},{"key":"1836_CR28","unstructured":"Alexey, D., Lucas, B., Alexander, K., Dirk, W., Xiaohua, Z., Thomas, U., Mostafa, D., Matthias, M., et al. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International conference on learning representations, (2021)"},{"key":"1836_CR29","unstructured":"Anurag, A., Mostafa, D., Georg, H., Chen, S., Mario, L., Cordelia, S. ViViT: A video vision transformer. In international conference on computer vision (2021)"},{"key":"1836_CR30","unstructured":"Haoqi, F., Bo, X., Karttikeya, M., Yanghao, L., Zhicheng, Y., Jitendra, M, Christoph, F.: Multiscale vision transformers. In international conference on computer vision (2021)"},{"key":"1836_CR31","unstructured":"Kunchang, L., Yali, W., Peng, G., Guanglu, S., Yu, L., Hongsheng, L., Yu, Q.: UniFormer: unified transformer for efficient spatiotemporal representation learning. In International conference on learning representations (2022)"},{"key":"1836_CR32","unstructured":"Zhan, T., Yibing, S., Jue, W., Limin, W.: VideoMAE: masked autoencoders are data-efficient learners for self-supervised video pre-training. In conference and workshop on neural information processing systems (2022)."},{"key":"1836_CR33","unstructured":"Kunchang, L., Yali, W., Yizhuo, L., Yi, W., Yinan, H., LiMin, W., Yu, Q.: Unmasked teacher: towards training-efficient video foundation models. In international conference on computer vision (2023)"},{"key":"1836_CR34","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1007\/s00530-023-01251-2","volume":"30","author":"Hu Cui","year":"2024","unstructured":"Cui, Hu., Hayama, Tessai: T. STSD: spatial\u2013temporal semantic decomposition transformer for skeleton-based action recognition. Multimed. Syst 30, 43 (2024). https:\/\/doi.org\/10.1007\/s00530-023-01251-2","journal-title":"Multimed. Syst"},{"key":"1836_CR35","unstructured":"Shoufa, C., Chongjian, G., Zhan, T., Jiangliu, W., Yibing, S., Jue, W., Ping, L.: AdaptFormer: adapting vision transformers for scalable visual recognition. In Conference and workshop on neural information processing systems (2022)"},{"key":"1836_CR36","unstructured":"Actionclip Mengmeng, W., Jiazheng, X., Yong, L.: ActionCLIP: a new paradigm for video action recognition. In IEEE transactions on neural networks and learning systems (2023)."},{"key":"1836_CR37","unstructured":"Hanoona, R., Muhammad, U. K., Muhammad, M., Salman, K., Fahad, S. K.: Fine-tuned CLIP Models are efficient video learners. In IEEE conference on computer vision and pattern recognition (2023)"},{"key":"1836_CR38","unstructured":"Zejia, W., Xitong, Y., Ang, L., Zuxuan, W., Yu-Gang, J.: Open-VCLIP: Transforming CLIP to an Open-vocabulary video model via interpolated weight optimization. In International conference on machine learning (2023)."},{"key":"1836_CR39","unstructured":"Geoffrey, H., Oriol, V., Jeff, D.: Distilling the knowledge in a neural network. In Conference and workshop on neural information processing systems (2014)"},{"key":"1836_CR40","unstructured":"Zheng, L., Ying, H., Defang, C., Tianren, L., Ning, C., Zhigeng, P.: Online knowledge distillation via multi-branch diversity enhancement. In Asian Conference on computer vision (2020)"},{"key":"1836_CR41","unstructured":"Borui, Z., Quan, C., Renjie, S., Yiyu, Q., Jiajun, L.: Decoupled knowledge distillation. In IEEE Conference on computer vision and pattern recognition (2022)"},{"key":"1836_CR42","unstructured":"Zheng, L., Xiang, L., Lingfeng, Y., Borui, Z., Renjie, S., Lei, L., Jun, L., Jian, Y. Curriculum temperature for knowledge distillation. In Association for the Advancement of artificial intelligence (2023)"},{"key":"1836_CR43","unstructured":"Ying, Z., Tao, X., Timothy, M. H., Huchuan, L.: Deep mutual learning. In IEEE Conference on computer vision and pattern recognition, (2018)"},{"key":"1836_CR44","unstructured":"Wonpyo, P., Dongju, K., Yan, L., and Minsu, C.: Relational knowledge distillation. In IEEE Conference on Computer Vision and Pattern Recognition (2019)"},{"key":"1836_CR45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20211","author":"Chuanguang Yang","year":"2022","unstructured":"Yang, Chuanguang, An, Zhulin, Cai, Linhang, Yongjun, Xu.: Mutual contrastive learning for visual representation Learning. In Assoc. Adv. Artificial Intell. (2022). https:\/\/doi.org\/10.1609\/aaai.v36i3.20211","journal-title":"In Assoc. Adv. Artificial Intell."},{"key":"1836_CR46","unstructured":"Defang, C., Jian-Ping, M., Hailin, Z., Can, W., Yan, F., Chun, C.: Knowledge distillation with the reused teacher classifier. In IEEE Conference on computer vision and pattern recognition (2022)"},{"key":"1836_CR47","unstructured":"Jing, Y., Brais, M., Adrian, B., Georgios, T., et al. Knowledge distillation via softmax regression representation learning. In: International Conference on Learning Representations (2021)"},{"key":"1836_CR48","unstructured":"Chen, J., Tengda, H., Kunhao, Z., Ya, Z., Weidi, X.: Prompting visual-language models for efficient video understanding. In: European conference on computer vision (2022)"},{"key":"1836_CR49","unstructured":"Gedas, B., Heng, W., Lorenzo T.: Is space-time attention all you need for video understanding? In international conference on machine learning, 2021."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01836-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01836-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01836-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T09:01:27Z","timestamp":1757926887000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01836-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,16]]},"references-count":49,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1836"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01836-z","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-5330691\/v1","asserted-by":"object"}]},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,16]]},"assertion":[{"value":"25 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 April 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"256"}}