{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T11:45:49Z","timestamp":1778931949471,"version":"3.51.4"},"reference-count":79,"publisher":"Springer Science and Business Media LLC","issue":"22","license":[{"start":{"date-parts":[[2024,8,13]],"date-time":"2024-08-13T00:00:00Z","timestamp":1723507200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,13]],"date-time":"2024-08-13T00:00:00Z","timestamp":1723507200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the National Key Research and Development Program of China","award":["No.2021ZD0111405"],"award-info":[{"award-number":["No.2021ZD0111405"]}]},{"name":"the Key Research and Development Program of Gansu Province","award":["No.21YF5GA103"],"award-info":[{"award-number":["No.21YF5GA103"]}]},{"name":"the Key Research and Development Program of Gansu Province","award":["No.21YF5FA111"],"award-info":[{"award-number":["No.21YF5FA111"]}]},{"name":"Lanzhou Science and Technology Planning Project","award":["No.2021-1-183"],"award-info":[{"award-number":["No.2021-1-183"]}]},{"DOI":"10.13039\/501100017702","name":"Innovation and Entrepreneurship Talent Project of Lanzhou","doi-asserted-by":"publisher","award":["No.2021-RC-91"],"award-info":[{"award-number":["No.2021-RC-91"]}],"id":[{"id":"10.13039\/501100017702","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1007\/s10489-024-05617-5","type":"journal-article","created":{"date-parts":[[2024,8,13]],"date-time":"2024-08-13T01:02:10Z","timestamp":1723510930000},"page":"11196-11211","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Cross-modal guides spatio-temporal enrichment network for few-shot action recognition"],"prefix":"10.1007","volume":"54","author":[{"given":"Zhiwen","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Min","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,13]]},"reference":[{"key":"5617_CR1","doi-asserted-by":"crossref","unstructured":"Ahn D, Kim S, Ko BC (2023) Star++: Rethinking spatio-temporal cross attention transformer for video action recognition. Appl Intell 1\u201314","DOI":"10.1109\/WACV56688.2023.00333"},{"key":"5617_CR2","doi-asserted-by":"crossref","unstructured":"Feng F, Ming Y, Hu N, Zhou J (2023) See, move and hear: a local-to-global multi-modal interaction network for video action recognition. Appl Intell 1\u201320","DOI":"10.1007\/s10489-023-04497-5"},{"issue":"13","key":"5617_CR3","doi-asserted-by":"publisher","first-page":"17115","DOI":"10.1007\/s10489-022-04369-4","volume":"53","author":"Y Qin","year":"2023","unstructured":"Qin Y, Liu B (2023) Otde: optimal transport distribution enhancement for few-shot video recognition. Appl Intell 53(13):17115\u201317127","journal-title":"Appl Intell"},{"key":"5617_CR4","doi-asserted-by":"publisher","first-page":"264","DOI":"10.1016\/j.ins.2023.03.058","volume":"633","author":"S Qiu","year":"2023","unstructured":"Qiu S, Fan T, Jiang J, Wang Z, Wang Y, Xu J, Sun T, Jiang N (2023) A novel two-level interactive action recognition model based on inertial data fusion. Inf Sci 633:264\u2013279","journal-title":"Inf Sci"},{"key":"5617_CR5","doi-asserted-by":"crossref","unstructured":"Nasirihaghighi S, Ghamsarian N, Stefanics D, Schoeffmann K, Husslein H (2023) Action recognition in video recordings from gynecologic laparoscopy. In: 2023 IEEE 36th International symposium on computer-based medical systems (CBMS), pp 29\u201334","DOI":"10.1109\/CBMS58004.2023.00187"},{"issue":"1","key":"5617_CR6","doi-asserted-by":"publisher","first-page":"72","DOI":"10.18178\/joig.11.1.72-81","volume":"11","author":"MA Abdelrazik","year":"2023","unstructured":"Abdelrazik MA, Zekry A, Mohamed WA (2023) Efficient hybrid algorithm for human action recognition. J Image Graph 11(1):72\u201381","journal-title":"J Image Graph"},{"key":"5617_CR7","doi-asserted-by":"publisher","first-page":"110427","DOI":"10.1016\/j.patcog.2024.110427","volume":"151","author":"Z Wu","year":"2024","unstructured":"Wu Z, Ma N, Wang C, Xu C, Xu G, Li M (2024) Spatial-temporal hypergraph based on dual-stage attention network for multi-view data lightweight action recognition. Pattern Recognit 151:110427","journal-title":"Pattern Recognit"},{"key":"5617_CR8","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, Sun C, Lu\u010di\u0107 M, Schmid C (2021) Vivit: A video vision transformer. In: Proceedings of the IEEE\/CVF International conference on computer vision, pp 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"issue":"11","key":"5617_CR9","doi-asserted-by":"publisher","first-page":"4125","DOI":"10.1109\/TPAMI.2020.2991965","volume":"43","author":"D Damen","year":"2020","unstructured":"Damen D, Doughty H, Farinella GM, Fidler S, Furnari A, Kazakos E, Moltisanti D, Munro J, Perrett T, Price W et al (2020) The epic-kitchens dataset: collection, challenges and baselines. IEEE Trans Pattern Anal Mach Intell 43(11):4125\u20134141","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"6","key":"5617_CR10","doi-asserted-by":"publisher","first-page":"6659","DOI":"10.1109\/TPAMI.2021.3058606","volume":"45","author":"H Coskun","year":"2021","unstructured":"Coskun H, Zia MZ, Tekin B, Bogo F, Navab N, Tombari F, Sawhney HS (2021) Domain-specific priors and meta learning for few-shot first-person action recognition. IEEE Trans Pattern Anal Mach Intell 45(6):6659\u20136673","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5617_CR11","doi-asserted-by":"crossref","unstructured":"Xing J, Wang M, Liu Y, Mu B (2023) Revisiting the spatial and temporal modeling for few-shot action recognition. In: Proceedings of the AAAI conference on artificial intelligence, vol 37, pp 3001\u20133009","DOI":"10.1609\/aaai.v37i3.25403"},{"key":"5617_CR12","doi-asserted-by":"crossref","unstructured":"Wang X, Zhang S, Qing Z, Gao C, Zhang Y, Zhao D, Sang N (2023) Molo: Motion-augmented long-short contrastive learning for few-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18011\u201318021","DOI":"10.1109\/CVPR52729.2023.01727"},{"key":"5617_CR13","doi-asserted-by":"crossref","unstructured":"Wang X, Zhang S, Cen J, Gao C, Zhang Y, Zhao D, Sang N (2023) Clip-guided prototype modulating for few-shot action recognition. Int J Comput Vis 1\u201314","DOI":"10.1007\/s11263-023-01917-4"},{"key":"5617_CR14","doi-asserted-by":"crossref","unstructured":"Zhang H, Zhang L, Qi X, Li H, Torr PH, Koniusz P (2020) Few-shot action recognition with permutation-invariant attention. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part V 16, pp 525\u2013542","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"5617_CR15","doi-asserted-by":"crossref","unstructured":"Cao K, Ji J, Cao Z, Chang C-Y, Niebles JC (2020) Few-shot video classification via temporal alignment. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10618\u201310627","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"5617_CR16","doi-asserted-by":"crossref","unstructured":"Thatipelli A, Narayan S, Khan S, Anwer RM, Khan FS, Ghanem B (2022) Spatio-temporal relation modeling for few-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19958\u201319967","DOI":"10.1109\/CVPR52688.2022.01933"},{"key":"5617_CR17","doi-asserted-by":"crossref","unstructured":"Wang X, Ye W, Qi Z, Zhao X, Wang G, Shan Y, Wang H (2021) Semantic-guided relation propagation network for few-shot action recognition. In: Proceedings of the 29th ACM international conference on multimedia, pp 816\u2013825","DOI":"10.1145\/3474085.3475253"},{"key":"5617_CR18","doi-asserted-by":"crossref","unstructured":"Lin C-C, Lin K, Wang L, Liu Z, Li L (2022) Cross-modal representation learning for zero-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19978\u201319988","DOI":"10.1109\/CVPR52688.2022.01935"},{"key":"5617_CR19","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763"},{"issue":"9","key":"5617_CR20","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Learning to prompt for vision-language models. Int J Comput Vis 130(9):2337\u20132348","journal-title":"Int J Comput Vis"},{"key":"5617_CR21","doi-asserted-by":"crossref","unstructured":"Gao P, Geng S, Zhang R, Ma T, Fang R, Zhang Y, Li H, Qiao Y (2023) Clip-adapter: better vision-language models with feature adapters. Int J Comput Vis 1\u201315","DOI":"10.1007\/s11263-023-01891-x"},{"key":"5617_CR22","doi-asserted-by":"crossref","unstructured":"Wang Z, Lu Y, Li Q, Tao X, Guo Y, Gong M, Liu T (2022) Cris: Clip-driven referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11686\u201311695","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"5617_CR23","doi-asserted-by":"crossref","unstructured":"Chao Y-W, Vijayanarasimhan S, Seybold B, Ross DA, Deng J, Sukthankar R (2018) Rethinking the faster r-cnn architecture for temporal action localization. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1130\u20131139","DOI":"10.1109\/CVPR.2018.00124"},{"key":"5617_CR24","doi-asserted-by":"crossref","unstructured":"Perrett T, Masullo A, Burghardt T, Mirmehdi M, Damen D (2021) Temporal-relational crosstransformers for few-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 475\u2013484","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"5617_CR25","doi-asserted-by":"publisher","first-page":"24303","DOI":"10.1007\/s11042-021-10721-6","volume":"80","author":"M Haddad","year":"2021","unstructured":"Haddad M, Ghassab VK, Najar F, Bouguila N (2021) A statistical framework for few-shot action recognition. Multimed Tools Appl 80:24303\u201324318","journal-title":"Multimed Tools Appl"},{"key":"5617_CR26","doi-asserted-by":"crossref","unstructured":"Liu T, Ma Y, Yang W, Ji W, Wang R, Jiang P (2022) Spatial-temporal interaction learning based two-stream network for action recognition, 606:864\u2013876","DOI":"10.1016\/j.ins.2022.05.092"},{"key":"5617_CR27","doi-asserted-by":"publisher","first-page":"109884","DOI":"10.1016\/j.asoc.2022.109884","volume":"132","author":"M Zong","year":"2023","unstructured":"Zong M, Wang R, Ma Y, Ji W (2023) Spatial and temporal saliency based four-stream network with multi-task learning for action recognition. Appl Soft Comput 132:109884","journal-title":"Appl Soft Comput"},{"issue":"1","key":"5617_CR28","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1007\/s00371-020-02012-2","volume":"38","author":"SJ Berlin","year":"2022","unstructured":"Berlin SJ, John M (2022) Spiking neural network based on joint entropy of optical flow features for human action recognition. Vis Comput 38(1):223\u2013237","journal-title":"Vis Comput"},{"key":"5617_CR29","doi-asserted-by":"publisher","first-page":"4104","DOI":"10.1109\/TIP.2022.3180585","volume":"31","author":"Y Liu","year":"2022","unstructured":"Liu Y, Yuan J, Tu Z (2022) Motion-driven visual tempo learning for video-based action recognition. IEEE Trans Image Process 31:4104\u20134116","journal-title":"IEEE Trans Image Process"},{"issue":"3","key":"5617_CR30","doi-asserted-by":"publisher","first-page":"3528","DOI":"10.1007\/s11227-023-05611-7","volume":"80","author":"SB Khobdeh","year":"2024","unstructured":"Khobdeh SB, Yamaghani MR, Sareshkeh SK (2024) Basketball action recognition based on the combination of yolo and a deep fuzzy lstm network. J Supercomput 80(3):3528\u20133553","journal-title":"J Supercomput"},{"key":"5617_CR31","doi-asserted-by":"publisher","first-page":"428","DOI":"10.1016\/j.neucom.2020.03.111","volume":"407","author":"J Cai","year":"2020","unstructured":"Cai J, Hu J, Tang X, Hung T-Y, Tan Y-P (2020) Deep historical long short-term memory network for action recognition. Neurocomputing 407:428\u2013438","journal-title":"Neurocomputing"},{"key":"5617_CR32","doi-asserted-by":"publisher","first-page":"264","DOI":"10.1016\/j.ins.2023.03.058","volume":"633","author":"S Qiu","year":"2023","unstructured":"Qiu S, Fan T, Jiang J, Wang Z, Wang Y, Xu J, Sun T, Jiang N (2023) A novel two-level interactive action recognition model based on inertial data fusion. Inf Sci 633:264\u2013279","journal-title":"Inf Sci"},{"key":"5617_CR33","doi-asserted-by":"publisher","first-page":"126289","DOI":"10.1016\/j.neucom.2023.126289","volume":"545","author":"C Cao","year":"2023","unstructured":"Cao C, Lu Y, Zhang Y, Jiang D, Zhang Y (2023) Efficient spatiotemporal context modeling for action recognition. Neurocomputing 545:126289","journal-title":"Neurocomputing"},{"key":"5617_CR34","doi-asserted-by":"publisher","first-page":"110575","DOI":"10.1016\/j.asoc.2023.110575","volume":"145","author":"G Zhang","year":"2023","unstructured":"Zhang G, Wen S, Li J, Che H (2023) Fast 3d-graph convolutional networks for skeleton-based action recognition. Appl Soft Comput 145:110575","journal-title":"Appl Soft Comput"},{"issue":"5","key":"5617_CR35","doi-asserted-by":"publisher","first-page":"2816","DOI":"10.3390\/s23052816","volume":"23","author":"R Vrskova","year":"2023","unstructured":"Vrskova R, Kamencay P, Hudec R, Sykora P (2023) A new deep-learning method for human activity recognition. Sensors 23(5):2816","journal-title":"Sensors"},{"key":"5617_CR36","doi-asserted-by":"crossref","unstructured":"Li Y, Ji B, Shi X, Zhang J, Kang B, Wang L (2020) Tea: Temporal excitation and aggregation for action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 909\u2013918","DOI":"10.1109\/CVPR42600.2020.00099"},{"key":"5617_CR37","doi-asserted-by":"crossref","unstructured":"Liu Z, Luo D, Wang Y, Wang L, Tai Y, Wang C, Li J, Huang F, Lu T (2020) Teinet: Towards an efficient architecture for video recognition. In: Proceedings of the AAAI conference on artificial intelligence, vol 34, pp 11669\u201311676","DOI":"10.1609\/aaai.v34i07.6836"},{"key":"5617_CR38","doi-asserted-by":"crossref","unstructured":"Liu Z, Wang L, Wu W, Qian C, Lu T (2021) Tam: Temporal adaptive module for video recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 13708\u201313718","DOI":"10.1109\/ICCV48922.2021.01345"},{"key":"5617_CR39","doi-asserted-by":"crossref","unstructured":"Wu G, Xu Y, Li J, Shi Z, Liu X (2023) Imperceptible adversarial attack with multi-granular spatio-temporal attention for video action recognition. IEEE Internet Things J","DOI":"10.1109\/JIOT.2023.3280737"},{"issue":"2","key":"5617_CR40","doi-asserted-by":"publisher","first-page":"487","DOI":"10.1007\/s00530-022-00961-3","volume":"29","author":"A Zhou","year":"2023","unstructured":"Zhou A, Ma Y, Ji W, Zong M, Yang P, Wu M, Liu M (2023) Multi-head attention-based two-stream efficientnet for action recognition. Multimed Syst 29(2):487\u2013498","journal-title":"Multimed Syst"},{"key":"5617_CR41","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"issue":"10s","key":"5617_CR42","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3505244","volume":"54","author":"S Khan","year":"2022","unstructured":"Khan S, Naseer M, Hayat M, Zamir SW, Khan FS, Shah M (2022) Transformers in vision: a survey. ACM Comput Surv (CSUR) 54(10s):1\u201341","journal-title":"ACM Comput Surv (CSUR)"},{"key":"5617_CR43","doi-asserted-by":"publisher","first-page":"916","DOI":"10.7717\/peerj-cs.916","volume":"8","author":"H Zhao","year":"2022","unstructured":"Zhao H, Chen Z, Guo L, Han Z (2022) Video captioning based on vision transformer and reinforcement learning. Peer J Comput Sci 8:916","journal-title":"Peer J Comput Sci"},{"key":"5617_CR44","doi-asserted-by":"publisher","first-page":"109897","DOI":"10.1016\/j.patcog.2023.109897","volume":"145","author":"W Huang","year":"2024","unstructured":"Huang W, Deng Y, Hui S, Wu Y, Zhou S, Wang J (2024) Sparse self-attention transformer for image inpainting. Pattern Recognit 145:109897","journal-title":"Pattern Recognit"},{"key":"5617_CR45","doi-asserted-by":"publisher","first-page":"105431","DOI":"10.1016\/j.engappai.2022.105431","volume":"116","author":"Z Chang","year":"2022","unstructured":"Chang Z, Lu Y, Wang X, Ran X (2022) Mgnet: Mutual-guidance network for few-shot semantic segmentation. Eng Appl Artif Intell 116:105431","journal-title":"Eng Appl Artif Intell"},{"issue":"25","key":"5617_CR46","doi-asserted-by":"publisher","first-page":"18251","DOI":"10.1007\/s00521-023-08758-9","volume":"35","author":"Z Chang","year":"2023","unstructured":"Chang Z, Lu Y, Ran X, Gao X, Wang X (2023) Few-shot semantic segmentation: a review on recent approaches. Neural Comput Appl 35(25):18251\u201318275","journal-title":"Neural Comput Appl"},{"key":"5617_CR47","doi-asserted-by":"crossref","unstructured":"Kim C-L, Lee G-E, Choi Y-J, Kang J, Kim B-G (2024) Channel selective relation network for efficient few-shot facial expression recognition. In: 2024 IEEE International conference on consumer electronics (ICCE), pp 1\u20133","DOI":"10.1109\/ICCE59016.2024.10444505"},{"issue":"1","key":"5617_CR48","doi-asserted-by":"publisher","first-page":"58","DOI":"10.47672\/ejt.1473","volume":"7","author":"J Bharadiya","year":"2023","unstructured":"Bharadiya J (2023) A comprehensive survey of deep learning techniques natural language processing. Eur J Technol 7(1):58\u201366","journal-title":"Eur J Technol"},{"issue":"3","key":"5617_CR49","doi-asserted-by":"publisher","first-page":"103664","DOI":"10.1016\/j.ipm.2024.103664","volume":"61","author":"H Ran","year":"2024","unstructured":"Ran H, Li W, Li L, Tian S, Ning X, Tiwari P (2024) Learning optimal inter-class margin adaptively for few-shot class-incremental learning via neural collapse-based meta-learning. Inf Process Manage 61(3):103664","journal-title":"Inf Process Manage"},{"key":"5617_CR50","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1016\/j.neunet.2023.10.039","volume":"169","author":"S Tian","year":"2024","unstructured":"Tian S, Li L, Li W, Ran H, Ning X, Tiwari P (2024) A survey on few-shot class-incremental learning. Neural Netw 169:307\u2013324","journal-title":"Neural Netw"},{"issue":"22","key":"5617_CR51","doi-asserted-by":"publisher","first-page":"26603","DOI":"10.1007\/s10489-023-04937-2","volume":"53","author":"Z Chang","year":"2023","unstructured":"Chang Z, Lu Y, Ran X, Gao X, Zhao H (2023) Simple yet effective joint guidance learning for few-shot semantic segmentation. Appl Intell 53(22):26603\u201326621","journal-title":"Appl Intell"},{"key":"5617_CR52","doi-asserted-by":"publisher","first-page":"109170","DOI":"10.1016\/j.patcog.2022.109170","volume":"135","author":"X Huang","year":"2023","unstructured":"Huang X, Choi SH (2023) Sapenet: Self-attention based prototype enhancement network for few-shot learning. Pattern Recognit 135:109170","journal-title":"Pattern Recognit"},{"key":"5617_CR53","unstructured":"Xing C, Rostamzadeh N, Oreshkin B, O\u00a0Pinheiro PO (2019) Adaptive cross-modal few-shot learning. Adv Neural Inf Process Syst 32"},{"key":"5617_CR54","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.neunet.2023.01.019","volume":"163","author":"Q Li","year":"2023","unstructured":"Li Q, Xie X, Zhang J, Shi G (2023) Few-shot human-object interaction video recognition with transformers. Neural Netw 163:1\u20139","journal-title":"Neural Netw"},{"key":"5617_CR55","doi-asserted-by":"crossref","unstructured":"Elsken T, Staffler B, Metzen JH, Hutter F (2020) Meta-learning of neural architectures for few-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12365\u201312375","DOI":"10.1109\/CVPR42600.2020.01238"},{"key":"5617_CR56","unstructured":"Lee Y, Choi S (2018) Gradient-based meta-learning with learned layerwise metric and subspace. In: International conference on machine learning, pp 2927\u20132936"},{"issue":"13","key":"5617_CR57","doi-asserted-by":"publisher","first-page":"17115","DOI":"10.1007\/s10489-022-04369-4","volume":"53","author":"Y Qin","year":"2023","unstructured":"Qin Y, Liu B (2023) Otde: optimal transport distribution enhancement for few-shot video recognition. Appl Intell 53(13):17115\u201317127","journal-title":"Appl Intell"},{"key":"5617_CR58","doi-asserted-by":"crossref","unstructured":"Yang F, Wang R, Chen X (2022) Sega: Semantic guided attention on visual prototype for few-shot learning. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 1056\u20131066","DOI":"10.1109\/WACV51458.2022.00165"},{"key":"5617_CR59","doi-asserted-by":"crossref","unstructured":"Sung F, Yang Y, Zhang L, Xiang T, Torr PH, Hospedales TM (2018) Learning to compare: Relation network for few-shot learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1199\u20131208","DOI":"10.1109\/CVPR.2018.00131"},{"key":"5617_CR60","doi-asserted-by":"publisher","first-page":"122086","DOI":"10.1016\/j.eswa.2023.122086","volume":"238","author":"R Ma","year":"2024","unstructured":"Ma R, Wu H, Wang X, Wang W, Ma Y, Zhao L (2024) Multi-view semantic enhancement model for few-shot knowledge graph completion. Expert Syst Appl 238:122086","journal-title":"Expert Syst Appl"},{"issue":"9","key":"5617_CR61","doi-asserted-by":"publisher","first-page":"4594","DOI":"10.1109\/TIP.2019.2910052","volume":"28","author":"Z Chen","year":"2019","unstructured":"Chen Z, Fu Y, Zhang Y, Jiang Y-G, Xue X, Sigal L (2019) Multi-level semantic feature augmentation for one-shot learning. IEEE Trans Image Process 28(9):4594\u20134605","journal-title":"IEEE Trans Image Process"},{"key":"5617_CR62","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1016\/j.patcog.2018.03.006","volume":"80","author":"J Lu","year":"2018","unstructured":"Lu J, Li J, Yan Z, Mei F, Zhang C (2018) Attribute-based synthetic network (abs-net): Learning more from pseudo feature representations. Pattern Recognit 80:129\u2013142","journal-title":"Pattern Recognit"},{"key":"5617_CR63","doi-asserted-by":"crossref","unstructured":"Zhu L, Yang Y (2018) Compound memory networks for few-shot video classification. In: Proceedings of the european conference on computer vision (ECCV), pp 751\u2013766","DOI":"10.1007\/978-3-030-01234-2_46"},{"key":"5617_CR64","doi-asserted-by":"crossref","unstructured":"Wang X, Lu Y, Yu W, Pang Y, Wang H (2024) Few-shot action recognition via multi-view representation learning. IEEE Trans Circuits Syst Video Technol","DOI":"10.1109\/TCSVT.2024.3384875"},{"key":"5617_CR65","doi-asserted-by":"publisher","unstructured":"Wang X, Zhang S, Qing Z, Tang M, Zuo Z, Gao C, Jin R, Sang N (2022) Hybrid relation guided set matching for few-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19948\u201319957. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01932","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"5617_CR66","doi-asserted-by":"crossref","unstructured":"Wang X, Zhang S, Qing Z, Zuo Z, Gao C, Jin R, Sang N (2023) Hyrsm++: Hybrid relation guided temporal set matching for few-shot action recognition. Preprint at arXiv:2301.03330","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"5617_CR67","doi-asserted-by":"crossref","unstructured":"Li C, Zhang J, Wu S, Jin X, Shan S (2023) Hierarchical compositional representations for few-shot action recognition. Preprint at arXiv:2208.09424","DOI":"10.1016\/j.cviu.2023.103911"},{"key":"5617_CR68","unstructured":"Zhang Y, Gong K, Zhang K, Li H, Qiao Y, Ouyang W, Yue X (2023) Meta-transformer: A unified framework for multimodal learning. Preprint at arXiv:2307.10802"},{"key":"5617_CR69","doi-asserted-by":"crossref","unstructured":"Goyal R, Ebrahimi\u00a0Kahou S, Michalski V, Materzynska J, Westphal S, Kim H, Haenel V, Fruend I, Yianilos P, Mueller-Freitag M (2017) The something something video database for learning and evaluating visual common sense. In: Proceedings of the IEEE international conference on computer vision, pp 5842\u20135850","DOI":"10.1109\/ICCV.2017.622"},{"key":"5617_CR70","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"5617_CR71","unstructured":"Soomro K, Zamir AR, Shah M (2012) Ucf101: A dataset of 101 human actions classes from videos in the wild. Preprint at arXiv:1212.0402"},{"key":"5617_CR72","doi-asserted-by":"crossref","unstructured":"Kuehne H, Jhuang H, Garrote E, Poggio T, Serre T (2011) Hmdb: a large video database for human motion recognition. In: 2011 International conference on computer vision, pp 2556\u20132563","DOI":"10.1109\/ICCV.2011.6126543"},{"issue":"1","key":"5617_CR73","first-page":"273","volume":"44","author":"L Zhu","year":"2020","unstructured":"Zhu L, Yang Y (2020) Label independent memory for semi-supervised few-shot video classification. IEEE Trans Pattern Anal Mach Intell 44(1):273\u2013285","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5617_CR74","doi-asserted-by":"crossref","unstructured":"Wu J, Zhang T, Zhang Z, Wu F, Zhang Y (2022) Motion-modulated temporal fragment alignment network for few-shot action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9151\u20139160","DOI":"10.1109\/CVPR52688.2022.00894"},{"key":"5617_CR75","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00894","DOI":"10.1109\/CVPR52688.2022.00894"},{"key":"5617_CR76","doi-asserted-by":"crossref","unstructured":"Zheng S, Chen S, Jin Q (2022) Few-shot action recognition with hierarchical matching and contrastive learning. In: European conference on computer vision, pp 297\u2013313","DOI":"10.1007\/978-3-031-19772-7_18"},{"key":"5617_CR77","doi-asserted-by":"crossref","unstructured":"Li S, Liu H, Qian R, Li Y, See J, Fei M, Yu X, Lin W (2022) Ta2n: Two-stage action alignment network for few-shot action recognition. In: Proceedings of the AAAI conference on artificial intelligence, vol 36, pp 1404\u20131411","DOI":"10.1609\/aaai.v36i2.20029"},{"key":"5617_CR78","unstructured":"Liu H, Lin W, Chen T, Li Y, Li S, See J (2023) Few-shot action recognition via intra-and inter-video information maximization. Preprint at arXiv:2305.06114"},{"key":"5617_CR79","doi-asserted-by":"crossref","unstructured":"Xing J, Wang M, Ruan Y, Chen B, Guo Y, Mu B, Dai G, Wang J, Liu Y (2023) Boosting few-shot action recognition with graph-guided hybrid matching. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1740\u20131750","DOI":"10.1109\/ICCV51070.2023.00167"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05617-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05617-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05617-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,18]],"date-time":"2024-09-18T15:22:42Z","timestamp":1726672962000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05617-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,13]]},"references-count":79,"journal-issue":{"issue":"22","published-print":{"date-parts":[[2024,11]]}},"alternative-id":["5617"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05617-5","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,13]]},"assertion":[{"value":"14 June 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}},{"value":"The authors declare that they have an informed consent to publish and for data used. This research is not napplicable for both human and\/or animal.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and informed consent for data used"}}]}}