{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T04:59:42Z","timestamp":1777870782691,"version":"3.51.4"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100016510","name":"Center for Autonomous Systems and Technologies","doi-asserted-by":"publisher","award":["2023QNRC001"],"award-info":[{"award-number":["2023QNRC001"]}],"id":[{"id":"10.13039\/100016510","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100016510","name":"Center for Autonomous Systems and Technologies","doi-asserted-by":"publisher","award":["24NLTSZ003"],"award-info":[{"award-number":["24NLTSZ003"]}],"id":[{"id":"10.13039\/100016510","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576279"],"award-info":[{"award-number":["62576279"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62301434"],"award-info":[{"award-number":["62301434"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376217"],"award-info":[{"award-number":["62376217"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.patcog.2026.113237","type":"journal-article","created":{"date-parts":[[2026,2,7]],"date-time":"2026-02-07T23:30:06Z","timestamp":1770507006000},"page":"113237","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Task-Adapter++: Task-specific adaptation with order-aware alignment for few-shot action recognition"],"prefix":"10.1016","volume":"177","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0217-9791","authenticated-orcid":false,"given":"Congqi","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9304-6580","authenticated-orcid":false,"given":"Peiheng","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9387-8764","authenticated-orcid":false,"given":"Yueran","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5504-9889","authenticated-orcid":false,"given":"Yating","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5359-5701","authenticated-orcid":false,"given":"Qinyi","family":"Lv","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3970-7823","authenticated-orcid":false,"given":"Lingtong","family":"Min","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2977-8057","authenticated-orcid":false,"given":"Yanning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113237_bib0001","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113237_bib0002","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.patcog.2026.113237_bib0003","doi-asserted-by":"crossref","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","article-title":"CLIP-adapter: better vision-language models with feature adapters","volume":"132","author":"Gao","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113237_bib0004","series-title":"The Eleventh International Conference on Learning Representations","article-title":"AIM: adapting image models for efficient video action recognition","author":"Yang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113237_bib0005","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6299","article-title":"Quo vadis, action recognition? a new model and the kinetics dataset","author":"Carreira","year":"2017"},{"key":"10.1016\/j.patcog.2026.113237_bib0006","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"5842","article-title":"The \u201csomething something\u201d video database for learning and evaluating visual common sense","author":"Goyal","year":"2017"},{"key":"10.1016\/j.patcog.2026.113237_bib0007","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8808","article-title":"Few-shot learning via embedding adaptation with set-to-set functions","author":"Ye","year":"2020"},{"key":"10.1016\/j.patcog.2026.113237_bib0008","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10618","article-title":"Few-shot video classification via temporal alignment","author":"Cao","year":"2020"},{"key":"10.1016\/j.patcog.2026.113237_bib0009","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"475","article-title":"Temporal-relational crosstransformers for few-shot action recognition","author":"Perrett","year":"2021"},{"key":"10.1016\/j.patcog.2026.113237_bib0010","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"19948","article-title":"Hybrid relation guided set matching for few-shot action recognition","author":"Wang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113237_bib0011","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"18218","article-title":"Frame order matters: a temporal sequence-aware model for few-Shot action recognition","volume":"39","author":"Li","year":"2025"},{"key":"10.1016\/j.patcog.2026.113237_bib0012","article-title":"MA-FSAR: Multimodal adaptation of CLIP for few-shot action recognition","volume":"169","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0013","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111283","article-title":"Text generation and multi-modal knowledge transfer for few-shot object detection","volume":"161","author":"Du","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0014","series-title":"International Conference on Machine Learning","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","author":"Houlsby","year":"2019"},{"key":"10.1016\/j.patcog.2026.113237_bib0015","series-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing","first-page":"3045","article-title":"The power of scale for parameter-efficient prompt tuning","author":"Lester","year":"2021"},{"key":"10.1016\/j.patcog.2026.113237_bib0016","series-title":"Proceedings of the 30th ACM International Conference on Multimedia","first-page":"6230","article-title":"Task-adaptive spatial-temporal video sampler for few-shot action recognition","author":"Liu","year":"2022"},{"key":"10.1016\/j.patcog.2026.113237_bib0017","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia, MM \u201924","first-page":"9038-9047","article-title":"Task-adapter: task-specific adaptation of image models for few-shot action recognition","author":"Cao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113237_bib0018","unstructured":"A. Mehrotra, A. Dukkipati, Generative adversarial residual pairwise networks for one shot learning, (2017). arXiv: 1703.08033."},{"key":"10.1016\/j.patcog.2026.113237_bib0019","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"7278","article-title":"Low-shot learning from imaginary data","author":"Wang","year":"2018"},{"key":"10.1016\/j.patcog.2026.113237_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111122","article-title":"PRSN: prototype resynthesis network with cross-image semantic alignment for few-shot image classification","volume":"159","author":"Dong","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110736","article-title":"A simple scheme to amplify inter-class discrepancy for improving few-shot fine-grained image classification","volume":"156","author":"Li","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0022","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110790","article-title":"Meta-collaborative comparison for effective cross-domain few-shot learning","volume":"156","author":"Zhou","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0023","series-title":"International Conference on Machine Learning","first-page":"1126","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","author":"Finn","year":"2017"},{"key":"10.1016\/j.patcog.2026.113237_bib0024","series-title":"International Conference on Machine Learning","first-page":"10617","article-title":"Metafun: meta-learning with iterative functional updates","author":"Xu","year":"2020"},{"key":"10.1016\/j.patcog.2026.113237_bib0025","first-page":"7957","article-title":"Fast and flexible multi-task classification using conditional neural adaptive processes","volume":"32","author":"Requeima","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113237_bib0026","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14493","article-title":"Improved few-shot visual classification","author":"Bateni","year":"2020"},{"key":"10.1016\/j.patcog.2026.113237_bib0027","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110559","article-title":"Ta-Adapter: enhancing few-shot CLIP with task-aware encoders","volume":"153","author":"Zhang","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0028","unstructured":"S. Lu, H.-J. Ye, D.-C. Zhan, Few-shot action recognition with compromised metric via optimal transport, (2021). arXiv: 2104.03737."},{"key":"10.1016\/j.patcog.2026.113237_bib0029","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"19958","article-title":"Spatio-temporal relation modeling for few-shot action recognition","author":"Thatipelli","year":"2022"},{"key":"10.1016\/j.patcog.2026.113237_bib0030","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"2243","article-title":"On the importance of spatial relations for few-shot action recognition","author":"Zhang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113237_bib0031","series-title":"Proceedings of the 29th ACM International Conference on Multimedia","first-page":"816","article-title":"Semantic-guided relation propagation network for few-shot action recognition","author":"Wang","year":"2021"},{"key":"10.1016\/j.patcog.2026.113237_bib0032","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6492","article-title":"Active exploration of multimodal complementarity for few-shot action recognition","author":"Wanyan","year":"2023"},{"key":"10.1016\/j.patcog.2026.113237_bib0033","doi-asserted-by":"crossref","first-page":"12349","DOI":"10.1109\/TNNLS.2024.3443394","article-title":"Enhancing few-shot CLIP with semantic-aware fine-tuning","volume":"36","author":"Zhu","year":"2024","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.113237_bib0034","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"751","article-title":"Compound memory networks for few-shot video classification","author":"Zhu","year":"2018"},{"key":"10.1016\/j.patcog.2026.113237_bib0035","series-title":"CVPR","article-title":"Spatio-temporal relation modeling for few-shot action recognition","author":"Thatipelli","year":"2022"},{"key":"10.1016\/j.patcog.2026.113237_bib0036","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18011","article-title":"MoLo: motion-augmented long-short contrastive learning for few-shot action recognition","author":"Wang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113237_bib0037","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.110110","article-title":"HyRSM++: hybrid relation guided temporal set matching for few-shot action recognition","volume":"147","author":"Wang","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113237_bib0038","doi-asserted-by":"crossref","first-page":"1899","DOI":"10.1007\/s11263-023-01917-4","article-title":"CLIP-Guided prototype modulating for few-shot action recognition","volume":"132","author":"Wang","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113237_bib0039","unstructured":"H. Qu, L. Xing, J. Zhang, R. Yan, Y. Yao, X. Shu, Hierarchical relation-augmented representation generalization for few-shot action recognition, (2025). arXiv: 2504.10079."},{"key":"10.1016\/j.patcog.2026.113237_bib0040","series-title":"International Conference on Computer Vision","first-page":"2556","article-title":"HMDB: ae video database for human motion recognition","author":"Kuehne","year":"2011"},{"key":"10.1016\/j.patcog.2026.113237_bib0041","series-title":"Proceedings of the First International Workshop on Action Recognition with Large Number of Classes, ICCV Workshops","article-title":"UCF101: A dataset of 101 human action classes from videos in the wild","author":"Soomro","year":"2013"},{"key":"10.1016\/j.patcog.2026.113237_bib0042","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part V 16","first-page":"525","article-title":"Few-shot action recognition with permutation-invariant attention","author":"Zhang","year":"2020"},{"issue":"1","key":"10.1016\/j.patcog.2026.113237_bib0043","first-page":"273","article-title":"Label independent memory for semi-supervised few-shot video classification","volume":"44","author":"Zhu","year":"2020","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113237_bib0044","unstructured":"OpenAI, J. Achiam, S. Adler, S. Agarwal, L. Ahmad, GPT-4 Technical Report, 2024."},{"key":"10.1016\/j.patcog.2026.113237_bib0045","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"2344","article-title":"Multi-speed global contextual subspace matching for few-shot action recognition","author":"Yu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113237_bib0046","unstructured":"A. Yang, B. Yang, B. Hui, Qwen2 Technical Report, 2024."},{"key":"10.1016\/j.patcog.2026.113237_bib0047","unstructured":"A. Grattafiori, A. Dubey, A. Jauhri, The Llama 3 Herd of Models, 2024."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326002025?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326002025?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:41:01Z","timestamp":1777567261000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326002025"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":47,"alternative-id":["S0031320326002025"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113237","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Task-Adapter++: Task-specific adaptation with order-aware alignment for few-shot action recognition","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113237","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113237"}}