{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T16:21:01Z","timestamp":1771258861557,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":21,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819561223","type":"print"},{"value":"9789819561230","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-6123-0_57","type":"book-chapter","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T15:43:41Z","timestamp":1771256621000},"page":"619-628","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multimodal Higher-Order Statistical Adapter For Video Action Recognition"],"prefix":"10.1007","author":[{"given":"Meng","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bingbing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,17]]},"reference":[{"key":"57_CR1","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision (2021)"},{"key":"57_CR2","unstructured":"Wang, M., Xing, J., Mei, J., Liu, Y., Jiang, Y.: ActionCLIP: adapting language-image pretrained models for video action recognition. IEEE Trans. Neural Netw. Learn. Syst. 2023"},{"key":"57_CR3","doi-asserted-by":"crossref","unstructured":"Ni, B., et al.: Expanding language-image pretrained models for general video recognition (2022)","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"57_CR4","doi-asserted-by":"publisher","unstructured":"Houlsby, N.: Parameter-efficient transfer learning for NLP\u2019, arXiv e-prints\u00a1\/, Art. no. arXiv. 1902.00751 (2019). https:\/\/doi.org\/10.48550\/arXiv.1902.00751.","DOI":"10.48550\/arXiv.1902.00751."},{"key":"57_CR5","doi-asserted-by":"publisher","unstructured":"Houlsby, N.: \u201cParameter-Efficient Transfer Learning for NLP\u201d, arXiv e-prints. Art. no. arXiv. 00751, 2019 (1902). https:\/\/doi.org\/10.48550\/arXiv.1902.00751","DOI":"10.48550\/arXiv.1902.00751"},{"key":"57_CR6","doi-asserted-by":"publisher","unstructured":"Chen, S., et al.: AdaptFormer: adapting vision transformers for scalable visual recognition (2022). https:\/\/doi.org\/10.48550\/arXiv.2205.13535","DOI":"10.48550\/arXiv.2205.13535"},{"key":"57_CR7","doi-asserted-by":"publisher","unstructured":"Wasim, S.T., Naseer, M., Khan, S., Khan, F.S., Shah, M.: Vita-CLIP: video and text adaptive CLIP via multimodal prompting (2023). https:\/\/doi.org\/10.48550\/arXiv.2304.03307.","DOI":"10.48550\/arXiv.2304.03307."},{"key":"57_CR8","unstructured":"Defferrard, M., Bresson, X., Vandergheynst, P.: Convolutional neural networks on graphs with fast localized spectral filtering. 10.48550\/arXiv.1606.09375 (2016)"},{"key":"57_CR9","doi-asserted-by":"publisher","unstructured":"Lin, Z., et al.: Frozen CLIP models are efficient video learners (2022). https:\/\/doi.org\/10.48550\/arXiv.2208.03550","DOI":"10.48550\/arXiv.2208.03550"},{"key":"57_CR10","doi-asserted-by":"publisher","unstructured":"Pan, J., Lin, Z., Zhu, X., Shao, J., Li, H.: ST-Adapter: Parameter-efficient image-to-video transfer learning (2022). https:\/\/doi.org\/10.48550\/arXiv.2206.13559","DOI":"10.48550\/arXiv.2206.13559"},{"key":"57_CR11","doi-asserted-by":"publisher","unstructured":"Wang, G., Gupta,H.: Non-local neural networks (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00813","DOI":"10.1109\/CVPR.2018.00813"},{"key":"57_CR12","unstructured":"Gao, Z., Wang, Q., Zhang, B., Hu, Q., Li, P.: Temporal-attentive covariance pooling networks for video recognition. In: Advances in Neural Information Processing Systems, vol. 34, pp. 13587\u201313598 (2021)"},{"key":"57_CR13","doi-asserted-by":"crossref","unstructured":"Gao, Z., Xie, J., Wang, Q., Li, P.: Global second-order pooling convolutional networks (2018)","DOI":"10.1109\/CVPR.2019.00314"},{"key":"57_CR14","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/j.aej.2024.11.067","volume":"114","author":"B Zhang","year":"2025","unstructured":"Zhang, B., Dong, W., Wang, Z., Zhang, J., Sun, Q.: Second-order transformer network for video recognition. Alexandria Eng. J. 114, 82\u201394 (2025). https:\/\/doi.org\/10.1016\/j.aej.2024.11.067","journal-title":"Alexandria Eng. J."},{"issue":"6","key":"57_CR15","doi-asserted-by":"publisher","first-page":"3866","DOI":"10.1109\/TCSVT.2021.3119983","volume":"32","author":"S Bai","year":"2022","unstructured":"Bai, S., Ma, B., Chang, H., Huang, R., Shan, S., Chen, X.: SANet: statistic attention network for video-based person re-identification. IEEE Trans. Circ. Syst. Video Technol. 32(6), 3866\u20133879 (2022). https:\/\/doi.org\/10.1109\/TCSVT.2021.3119983","journal-title":"IEEE Trans. Circ. Syst. Video Technol."},{"key":"57_CR16","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer, Hierarchical vision transformer using shifted windows. arXiv (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"57_CR17","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C-Y., Fan, H., Mangalam, K., Xiong, B., Malik, J., Feichtenhofer, C.: MViTv2: Improved multiscale vision transformers for classification and detection (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"57_CR18","doi-asserted-by":"crossref","unstructured":"Wu, W., Wang, X., Luo, H., Wang, J., Yang, Y., Ouyang, W.: Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models (2023)","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"57_CR19","doi-asserted-by":"publisher","unstructured":"Yang, T., Zhu, Y., Xie, Y., Zhang, A., Chen, C., Li, M.: AIM: Adapting Image Models For Efficient Video Action Recognition. arXiv (2023) https:\/\/doi.org\/10.48550\/arXiv.2302.03024.","DOI":"10.48550\/arXiv.2302.03024."},{"key":"57_CR20","doi-asserted-by":"publisher","unstructured":"Wang, M., et al.: M2-CLIP: a multimodal, multi-task adapting framework for video action recognition (2024). https:\/\/doi.org\/10.48550\/arXiv.2401.11649.","DOI":"10.48550\/arXiv.2401.11649."},{"issue":"09","key":"57_CR21","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vis. 130(09), 2337\u20132348 (2022). https:\/\/doi.org\/10.1007\/s11263-022-01653-1","journal-title":"Int. J. Comput. Vis."}],"container-title":["Lecture Notes in Computer Science","Biometric Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-6123-0_57","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T15:43:44Z","timestamp":1771256624000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-6123-0_57"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819561223","9789819561230"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-6123-0_57","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"17 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CCBR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Biometric Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Nanchang","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccbr2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ccbr99.cn\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}