{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T18:30:50Z","timestamp":1768329050396,"version":"3.49.0"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819556755","type":"print"},{"value":"9789819556762","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-5676-2_18","type":"book-chapter","created":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T12:30:13Z","timestamp":1768307413000},"page":"262-276","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["High-Order Multimodal Multi-task Video Action Recognition"],"prefix":"10.1007","author":[{"given":"Bingbing","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongqi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,13]]},"reference":[{"key":"18_CR1","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. PMLR, vol. 139, pp. 8748\u20138763 (2021)"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Tu, S., Dai, Q., Wu, Z., Cheng, Z.-Q., Hu, H., Jiang, Y.-G.: Implicit temporal modeling with learnable alignment for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 19936\u201319947 (2023)","DOI":"10.1109\/ICCV51070.2023.01825"},{"issue":"1","key":"18_CR3","doi-asserted-by":"publisher","first-page":"625","DOI":"10.1109\/TNNLS.2023.3331841","volume":"36","author":"M Wang","year":"2025","unstructured":"Wang, M., Xing, J., Mei, J., Liu, Y., Jiang, Y.: ActionCLIP: adapting language-image pretrained models for video action recognition. IEEE Trans. Neural Netw. Learn. Syst. 36(1), 625\u2013637 (2025). https:\/\/doi.org\/10.1109\/TNNLS.2023.3331841","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"18_CR4","unstructured":"Han, Z., Gao, C., Liu, J., Zhang, J., Zhang, S.Q.: Parameter-efficient fine-tuning for large models: a comprehensive survey. arXiv preprint arXiv:2403.14608 (2024)"},{"key":"18_CR5","first-page":"26462","volume":"35","author":"J Pan","year":"2022","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., Li, H.: ST-Adapter: parameter-efficient image-to-video transfer learning. Adv. Neural. Inf. Process. Syst. 35, 26462\u201326477 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"2","key":"18_CR6","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao, P., et al.: Clip-adapter: better vision-language models with feature adapters. Int. J. Comput. Vision 132(2), 581\u2013595 (2024)","journal-title":"Int. J. Comput. Vision"},{"issue":"9","key":"18_CR7","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16816\u201316825 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Jia, M., et al.: Visual prompt tuning. In: European Conference on Computer Vision, pp. 709\u2013727. Springer (2022)","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., Shah, M.: Vita-clip: video and text adaptive clip via multimodal prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23034\u201323044 (2023)","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Xie, S., Sun, C., Huang, J., Tu, Z., Murphy, K.: Rethinking spatiotemporal feature learning: Speed-accuracy trade-offs in video classification. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 305\u2013321 (2018)","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"18_CR13","unstructured":"Soomro, K.: UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., Serre, T.: HMDB: a large video database for human motion recognition. In: 2011 International Conference on Computer Vision, pp. 2556\u20132563. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"issue":"3","key":"18_CR15","doi-asserted-by":"publisher","first-page":"220","DOI":"10.1038\/s42256-023-00626-4","volume":"5","author":"N Ding","year":"2023","unstructured":"Ding, N., et al.: Parameter-efficient fine-tuning of large-scale pre-trained language models. Nat. Mach. Intell. 5(3), 220\u2013235 (2023)","journal-title":"Nat. Mach. Intell."},{"key":"18_CR16","unstructured":"Yang, T., Zhu, Y., Xie, Y., Zhang, A., Chen, C., Li, M.: AIM: adapting image models for efficient video action recognition. arXiv preprint arXiv:2302.03024 (2023)"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Gao, Z., Xie, J., Wang, Q., Li, P.: Global second-order pooling convolutional networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3024\u20133033 (2019)","DOI":"10.1109\/CVPR.2019.00314"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., Khan, F.S.: Maple: multi-modal prompt learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19113\u201319122 (2023)","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., Shah, M.: Vita-CLIP: video and text adaptive CLIP via multimodal prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23034\u201323044 (2023)","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"18_CR20","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Li, P., Xie, J., Wang, Q., Gao, Z.: Towards faster training of global covariance pooling networks by iterative matrix square root normalization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 947\u2013955 (2018)","DOI":"10.1109\/CVPR.2018.00105"},{"key":"18_CR22","first-page":"13587","volume":"34","author":"Z Gao","year":"2021","unstructured":"Gao, Z., Wang, Q., Zhang, B., Hu, Q., Li, P.: Temporal-attentive covariance pooling networks for video recognition. Adv. Neural. Inf. Process. Syst. 34, 13587\u201313598 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"18_CR23","doi-asserted-by":"crossref","unstructured":"Liang, R., Yan, L., Gao, P., Qian, X., Zhang, Z., Sun, H.: Aviation video moving-target detection with inter-frame difference. In: 2010 3rd International Congress on Image and Signal Processing, vol. 3, pp. 1494\u20131497 (2010)","DOI":"10.1109\/CISP.2010.5646303"},{"key":"18_CR24","unstructured":"Wang, M., et al.: M2-CLIP: a multimodal, multi-task adapting framework for video action recognition. arXiv preprint arXiv:2401.11649 (2024)"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211 (2022)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: MViTv2: improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Tu, S., Dai, Q., Wu, Z., Cheng, Z.-Q., Hu, H., Jiang, Y.-G.: Implicit temporal modeling with learnable alignment for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19936\u201319947 (2023)","DOI":"10.1109\/ICCV51070.2023.01825"},{"key":"18_CR28","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Luo, C., Tang, C., Chen, D., Codella, N., Zha, Z.-J.: Streaming video model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14602\u201314612 (2023)","DOI":"10.1109\/CVPR52729.2023.01403"},{"key":"18_CR29","doi-asserted-by":"crossref","unstructured":"Ni, B., et al.: Expanding language-image pretrained models for general video recognition. In: European Conference on Computer Vision, pp. 1\u201318. Springer (2022)","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Wu, W., Wang, X., Luo, H., Wang, J., Yang, Y., Ouyang, W.: Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6620\u20136630 (2023)","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"18_CR31","unstructured":"Li, K., Wang, Y., He, Y., Li, Y., Wang, Y., Wang, L., Qiao, Y.: UniformerV2: spatiotemporal learning by arming image ViTs with video uniformer. arXiv preprint arXiv:2211.09552 (2022)"},{"key":"18_CR32","doi-asserted-by":"crossref","unstructured":"Lin, Z., et al.: Frozen CLIP models are efficient video learners. In: European Conference on Computer Vision, pp. 388\u2013404. Springer (2022)","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"18_CR33","doi-asserted-by":"crossref","unstructured":"Park, J., Lee, J., Sohn, K.: Dual-path adaptation from image to video transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2203\u20132213 (2023)","DOI":"10.1109\/CVPR52729.2023.00219"},{"key":"18_CR34","doi-asserted-by":"crossref","unstructured":"Liu, R., Huang, J., Li, G., Feng, J., Wu, X., Li, T.H.: Revisiting temporal modeling for CLIP-based image-to-video knowledge transferring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6555\u20136564 (2023)","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"18_CR35","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., Xie, W.: Prompting visual-language models for efficient video understanding. In: Computer Vision \u2013 ECCV 2022, pp. 105\u2013124. Springer Nature Switzerland (2022)","DOI":"10.1007\/978-3-031-19833-5_7"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-5676-2_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T12:30:21Z","timestamp":1768307421000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-5676-2_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819556755","9789819556762"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-5676-2_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"13 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Shanghai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 October 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2025.prcv.cn\/index.asp","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}