{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T20:09:49Z","timestamp":1785701389626,"version":"3.56.0"},"reference-count":87,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2023,10,17]],"date-time":"2023-10-17T00:00:00Z","timestamp":1697500800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,17]],"date-time":"2023-10-17T00:00:00Z","timestamp":1697500800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s11263-023-01917-4","type":"journal-article","created":{"date-parts":[[2023,10,17]],"date-time":"2023-10-17T06:02:45Z","timestamp":1697522565000},"page":"1899-1912","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":74,"title":["CLIP-guided Prototype Modulating for Few-shot Action Recognition"],"prefix":"10.1007","volume":"132","author":[{"given":"Xiang","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Cen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changxin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingya","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deli","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nong","family":"Sang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,17]]},"reference":[{"key":"1917_CR1","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., & Schmid, C. (2021). Vivit: A video vision transformer. In ICCV, pp. 6836\u20136846.","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"1917_CR2","unstructured":"Bishay, M., Zoumpourlis, G., & Patras, I. (2019). TARN: temporal attentive relation network for few-shot and zero-shot action recognition. In BMVC, BMVA Press, p. 154, https:\/\/bmvc2019.org\/wp-content\/uploads\/papers\/0650-paper.pdf"},{"key":"1917_CR3","doi-asserted-by":"crossref","unstructured":"Cao, K., Ji, J., Cao, Z., Chang, C.Y., & Niebles, J.C. (2020). Few-shot video classification via temporal alignment. In CVPR, pp. 10618\u201310627.","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"1917_CR4","doi-asserted-by":"crossref","unstructured":"Carreira, J., & Zisserman, A. (2017). Quo vadis, action recognition? a new model and the kinetics dataset. In CVPR, pp. 6299\u20136308.","DOI":"10.1109\/CVPR.2017.502"},{"key":"1917_CR5","doi-asserted-by":"crossref","unstructured":"Chen, C.F.R., Panda, R., Ramakrishnan, K., Feris, R., Cohn, J., Oliva, A., & Fan, Q. (2021). Deep analysis of cnn-based spatio-temporal representations for action recognition. In CVPR, pp. 6165\u20136175.","DOI":"10.1109\/CVPR46437.2021.00610"},{"key":"1917_CR6","unstructured":"Chung, J., Gulcehre, C., Cho, K., Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint arXiv:1412.3555"},{"key":"1917_CR7","doi-asserted-by":"crossref","unstructured":"Dai Z, Yang Z, Yang Y, Carbonell J, Le, Q.V., Salakhutdinov, R. (2019). Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860","DOI":"10.18653\/v1\/P19-1285"},{"key":"1917_CR8","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In CVPR, pp. 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1917_CR9","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., & Gelly, S., et\u00a0al. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In International Conference on Learning Representations."},{"issue":"4","key":"1917_CR10","doi-asserted-by":"publisher","first-page":"594","DOI":"10.1109\/TPAMI.2006.79","volume":"28","author":"L Fei-Fei","year":"2006","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2006). One-shot learning of object categories. IEEE Transactions on Pattern Analysis and Machine Intelligence, 28(4), 594\u2013611.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1917_CR11","unstructured":"Finn, C., Abbeel, P., & Levine, S. (2017). Model-agnostic meta-learning for fast adaptation of deep networks. In ICML, PMLR, pp. 1126\u20131135."},{"key":"1917_CR12","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., & Qiao, Y. (2021). Clip-adapter: Better vision-language models with feature adapters. arXiv preprint arXiv:2110.04544"},{"key":"1917_CR13","doi-asserted-by":"crossref","unstructured":"Goyal, R., Michalski, V., Materzy, J., Westphal, S., Kim, H., Haenel, V., Yianilos, P., Mueller-freitag, M., Hoppe, F., Thurau, C., Bax, I., & Memisevic, R. (2017). The \u201cSomething Something\u201d Video Database for Learning and Evaluating Visual Common Sense. In ICCV.","DOI":"10.1109\/ICCV.2017.622"},{"key":"1917_CR14","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, Ar., & Hinton, G. (2013). Speech recognition with deep recurrent neural networks. In ICASSP, pp. 6645\u20136649.","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"1917_CR15","unstructured":"Gu, X., Lin, T.Y., Kuo, W., Cui, Y. (2022). Open-vocabulary object detection via vision and language knowledge distillation. In ICLR."},{"key":"1917_CR16","doi-asserted-by":"crossref","unstructured":"Guo, Y., Codella, N.C., Karlinsky, L., Codella, J.V., Smith, J.R., Saenko, K., Rosing, T., & Feris, R. (2020). A broader study of cross-domain few-shot learning. In ECCV, Springer, pp. 124\u2013141.","DOI":"10.1007\/978-3-030-58583-9_8"},{"key":"1917_CR17","doi-asserted-by":"crossref","unstructured":"Hariharan, B., & Girshick, R. (2017). Low-shot visual recognition by shrinking and hallucinating features. In ICCV, pp. 3018\u20133027.","DOI":"10.1109\/ICCV.2017.328"},{"key":"1917_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In CVPR, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1917_CR19","doi-asserted-by":"crossref","unstructured":"Huang, Y., Yang, L., & Sato, Y. (2022). Compound prototype matching for few-shot action recognition. In ECCV.","DOI":"10.1007\/978-3-031-19772-7_21"},{"key":"1917_CR20","doi-asserted-by":"crossref","unstructured":"Jamal, M.A., & Qi, G.J. (2019). Task agnostic meta-learning for few-shot learning. In CVPR, pp. 11719\u201311727.","DOI":"10.1109\/CVPR.2019.01199"},{"key":"1917_CR21","unstructured":"Jia. C., Yang, Y., Xia, Y., Chen, Y.T., Parekh, Z., Pham, H., Le, Q., Sung, Y.H., Li, Z., & Duerig, T. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. In ICML, PMLR, pp. 4904\u20134916."},{"key":"1917_CR22","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., & Xie, W. (2022). Prompting visual-language models for efficient video understanding. In ECCV.","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"1917_CR23","unstructured":"Kingma, D.P., & Ba, J. (2014). Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"1917_CR24","doi-asserted-by":"publisher","unstructured":"Kuehne, H., Serre, T., Jhuang, H., Garrote, E., Poggio, T., & Serre, T. (2011). HMDB: A large video database for human motion recognition. In ICCV, https:\/\/doi.org\/10.1109\/ICCV.2011.6126543","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"1917_CR25","unstructured":"Li, B., Weinberger, K.Q., Belongie, S., Koltun, V., & Ranftl, R. (2022a). Language-driven semantic segmentation. In ICLR."},{"key":"1917_CR26","doi-asserted-by":"crossref","unstructured":"Li, H., Eigen, D., Dodge, S., Zeiler, M., & Wang, X. (2019). Finding task-relevant features for few-shot learning by category traversal. In CVPR, pp. 1\u201310.","DOI":"10.1109\/CVPR.2019.00009"},{"key":"1917_CR27","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C., & Hoi, S. C. H. (2021). Align before fuse: Vision and language representation learning with momentum distillation. NeurIPS, 34, 9694\u20139705.","journal-title":"NeurIPS"},{"key":"1917_CR28","doi-asserted-by":"crossref","unstructured":"Li, K., Zhang, Y., Li, K., & Fu, Y. (2020a). Adversarial feature hallucination networks for few-shot learning. In CVPR, pp. 13470\u201313479.","DOI":"10.1109\/CVPR42600.2020.01348"},{"key":"1917_CR29","doi-asserted-by":"crossref","unstructured":"Li, S., Liu, H., Qian, R., Li, Y., See, J., Fei, M., Yu, X., & Lin, W. (2022b). Ta2n: Two-stage action alignment network for few-shot action recognition. In AAAI, pp. 1404\u20131411.","DOI":"10.1609\/aaai.v36i2.20029"},{"key":"1917_CR30","doi-asserted-by":"crossref","unstructured":"Li, W., Gao, C., Niu, G., Xiao, X., Liu, H., Liu, J., Wu, H., & Wang, H. (2020b). Unimo: Towards unified-modal understanding and generation via cross-modal contrastive learning. arXiv preprint arXiv:2012.15409","DOI":"10.18653\/v1\/2021.acl-long.202"},{"key":"1917_CR31","doi-asserted-by":"crossref","unstructured":"Li, X., Yin, X., Li, C., Zhang, P., Hu, X., Zhang, L., Wang, L., Hu, H., Dong, L., & Wei, F., et\u00a0al. (2020c). Oscar: Object-semantics aligned pre-training for vision-language tasks. In ECCV, Springer, pp. 121\u2013137","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"1917_CR32","unstructured":"Li, Z., Zhou, F., Chen, F., & Li, H. (2017). Meta-sgd: Learning to learn quickly for few-shot learning. arXiv preprint arXiv:1707.09835"},{"key":"1917_CR33","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., & Han, S. (2019). Tsm: Temporal shift module for efficient video understanding. In ICCV, pp. 7083\u20137093","DOI":"10.1109\/ICCV.2019.00718"},{"key":"1917_CR34","doi-asserted-by":"crossref","unstructured":"Lin, Z., Geng, S., Zhang, R., Gao, P., de\u00a0Melo, G., Wang, X., Dai, J., Qiao, Y., & Li, H (2022). Frozen clip models are efficient video learners. In ECCV.","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"1917_CR35","doi-asserted-by":"crossref","unstructured":"Liu, Y., Xiong, P., Xu, L., Cao, S., & Jin, Q. (2022). Ts2-net: Token shift and selection transformer for text-video retrieval. In ECCV.","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"1917_CR36","doi-asserted-by":"crossref","unstructured":"Luo, J., Li, Y., Pan, Y., Yao, T., Chao, H., & Mei, T. (2021). Coco-bert: Improving video-language pre-training with contrastive cross-modal matching and denoising. In ACMMM, pp. 5600\u20135608.","DOI":"10.1145\/3474085.3475703"},{"key":"1917_CR37","doi-asserted-by":"crossref","unstructured":"M\u00fcller, M. (2007). Dynamic time warping. Information Retrieval for Music and Motion pp. 69\u201384.","DOI":"10.1007\/978-3-540-74048-3_4"},{"key":"1917_CR38","doi-asserted-by":"crossref","unstructured":"Nguyen, K.D., Tran, Q.H., Nguyen, K., Hua, B.S., & Nguyen, R. (2022). Inductive and transductive few-shot video classification via appearance and temporal alignments. In ECCV.","DOI":"10.1007\/978-3-031-20044-1_27"},{"key":"1917_CR39","doi-asserted-by":"crossref","unstructured":"Ni, B., Peng, H., Chen, M., Zhang, S., Meng, G., Fu, J., Xiang, S., & Ling, H. (2022). Expanding language-image pretrained models for general video recognition. In ECCV, Springer, pp. 1\u201318.","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"1917_CR40","doi-asserted-by":"crossref","unstructured":"Pahde, F., Ostapenko, O., Hnichen, P.J., Klein, T., & Nabi, M. (2019). Self-paced adversarial training for multimodal few-shot learning. In WACV, IEEE, pp. 218\u2013226.","DOI":"10.1109\/WACV.2019.00029"},{"key":"1917_CR41","doi-asserted-by":"crossref","unstructured":"Pahde, F., Puscas, M., Klein, T., & Nabi, M. (2021). Multimodal prototypical networks for few-shot learning. In WACV, pp. 2644\u20132653.","DOI":"10.1109\/WACV48630.2021.00269"},{"key":"1917_CR42","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., & Antiga, L., et\u00a0al. (2019). Pytorch: An imperative style, high-performance deep learning library. NeurIPS 32."},{"key":"1917_CR43","doi-asserted-by":"crossref","unstructured":"Perrett, T., Masullo, A., Burghardt, T., Mirmehdi, M., & Damen, D. (2021). Temporal-relational crosstransformers for few-shot action recognition. In CVPR, pp. 475\u2013484.","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"1917_CR44","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J., et\u00a0al. (2021). Learning transferable visual models from natural language supervision. In ICML, PMLR, pp. 8748\u20138763."},{"key":"1917_CR45","unstructured":"Rajeswaran, A., Finn, C., Kakade, S.M., & Levine, S. (2019). Meta-learning with implicit gradients. In NeurIPS, vol\u00a032."},{"key":"1917_CR46","doi-asserted-by":"crossref","unstructured":"Rao, Y., Zhao, W., Chen, G., Tang, Y., Zhu, Z., Huang, G., Zhou. J., & Lu, J. (2022). Denseclip: Language-guided dense prediction with context-aware prompting. In CVPR, pp. 18082\u201318091.","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"1917_CR47","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Khattak, M.U., Maaz, M., Khan, S., & Khan, F.S. (2023). Fine-tuned clip models are efficient video learners. In CVPR, pp. 6545\u20136554.","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"1917_CR48","unstructured":"Ravi, S., & Larochelle, H. (2017). Optimization as a model for few-shot learning. In ICLR."},{"key":"1917_CR49","unstructured":"Rusu, A.A., Rao, D., Sygnowski, J., Vinyals, O., Pascanu, R., Osindero, S., & Hadsell, R. (2019). Meta-learning with latent embedding optimization. In ICLR."},{"key":"1917_CR50","doi-asserted-by":"crossref","unstructured":"Shi, H., Hayat, M., Wu, Y., & Cai, J. (2022). Proposalclip: Unsupervised open-category object proposal generation via exploiting clip cues. In CVPR, pp. 9611\u20139620.","DOI":"10.1109\/CVPR52688.2022.00939"},{"key":"1917_CR51","doi-asserted-by":"crossref","unstructured":"Shi, Z., Liang, J., Li, Q., Zheng, H., Gu, Z., Dong, J., & Zheng, B. (2021). Multi-modal multi-action video recognition. In ICCV, pp. 13678\u201313687.","DOI":"10.1109\/ICCV48922.2021.01342"},{"key":"1917_CR52","first-page":"4077","volume":"30","author":"J Snell","year":"2017","unstructured":"Snell, J., Swersky, K., & Zemel, R. (2017). Prototypical networks for few-shot learning. NeurIPS, 30, 4077\u20134087.","journal-title":"NeurIPS"},{"key":"1917_CR53","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. arXiv arXiv:1212.0402"},{"key":"1917_CR54","doi-asserted-by":"crossref","unstructured":"Sung, F., Yang, Y., Zhang, L., Xiang, T., Torr, P,H., & Hospedales, T.M. (2018). Learning to compare: Relation network for few-shot learning. In CVPR, pp. 1199\u20131208.","DOI":"10.1109\/CVPR.2018.00131"},{"key":"1917_CR55","doi-asserted-by":"crossref","unstructured":"Thatipelli, A., Narayan, S., Khan, S., Anwer, R,M., Khan, F,S., & Ghanem, B. (2022). Spatio-temporal relation modeling for few-shot action recognition. In CVPR, pp. 19958\u201319967.","DOI":"10.1109\/CVPR52688.2022.01933"},{"key":"1917_CR56","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. In NeurIPS, pp. 5998\u20136008."},{"key":"1917_CR57","unstructured":"Vinyals, O., Blundell, C., Lillicrap, T., Kavukcuoglu, K., & Wierstra, D. (2016). Matching Networks for One Shot Learning. In: NeurIPS, arXiv:1606.04080v2"},{"key":"1917_CR58","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., & Van\u00a0Gool, L. (2016). Temporal segment networks: Towards good practices for deep action recognition. In ECCV, Springer, pp. 20\u201336.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1917_CR59","unstructured":"Wang, M., Xing, J., & Liu, Y. (2021a). Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:2109.08472"},{"key":"1917_CR60","unstructured":"Wang, T., Jiang, W., Lu, Z., Zheng, F., Cheng, R., Yin, C., & Luo, P. (2022a). Vlmixer: Unpaired vision-language pre-training via cross-modal cutmix. In ICML, PMLR, pp. 22680\u201322690."},{"key":"1917_CR61","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Shao, Y., Gao, C., & Sang, N. (2021b). Self-supervised learning for semi-supervised temporal action proposal. In CVPR, pp. 1905\u20131914.","DOI":"10.1109\/CVPR46437.2021.00194"},{"key":"1917_CR62","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Shao, Y., Zuo, Z., Gao, C., & Sang, N. (2021c). Oadtr: Online action detection with transformers. In ICCV, pp. 7565\u20137575.","DOI":"10.1109\/ICCV48922.2021.00747"},{"key":"1917_CR63","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Tang, M., Zuo, Z., Gao, C., Jin, R., & Sang, N. (2022b). Hybrid relation guided set matching for few-shot action recognition. In CVPR, pp. 19948\u201319957.","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"1917_CR64","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Zuo, Z., Gao, C., Jin, R., & Sang, N. (2023). Hyrsm++: Hybrid relation guided temporal set matching for few-shot action recognition. arXiv preprint arXiv:2301.03330","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"1917_CR65","doi-asserted-by":"crossref","unstructured":"Wang, Z., Lu, Y., Li, Q., Tao, X., Guo, Y., Gong, M., & Liu, T. (2022c). Cris: Clip-driven referring image segmentation. In CVPR, pp. 11686\u201311695.","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"1917_CR66","doi-asserted-by":"crossref","unstructured":"Wu, J., Zhang, T., Zhang, Z., Wu, .F, & Zhang, Y. (2022). Motion-modulated temporal fragment alignment network for few-shot action recognition. In CVPR, pp. 9151\u20139160.","DOI":"10.1109\/CVPR52688.2022.00894"},{"key":"1917_CR67","doi-asserted-by":"crossref","unstructured":"Wu, W., Sun, Z., & Ouyang, W. (2023). Revisiting classifier: Transferring vision-language models for video recognition. In AAAI, pp. 7\u20138.","DOI":"10.1609\/aaai.v37i3.25386"},{"key":"1917_CR68","unstructured":"Xing, C., Rostamzadeh, N., Oreshkin, B., & O\u00a0Pinheiro, P.O. (2019). Adaptive cross-modal few-shot learning. NeurIPS 32."},{"issue":"7","key":"1917_CR69","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1007\/s11263-022-01613-9","volume":"130","author":"W Xu","year":"2022","unstructured":"Xu, W., Xian, Y., Wang, J., Schiele, B., & Akata, Z. (2022). Attribute prototype network for any-shot learning. IJCV, 130(7), 1735\u20131753.","journal-title":"IJCV"},{"key":"1917_CR70","doi-asserted-by":"crossref","unstructured":"Yang, J., Li, C., Zhang, P., Xiao, B., Liu, C., Yuan, L., & Gao, J. (2022). Unified contrastive learning in image-text-label space. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19163\u201319173.","DOI":"10.1109\/CVPR52688.2022.01857"},{"key":"1917_CR71","doi-asserted-by":"crossref","unstructured":"Ye, H.J., Hu, H., Zhan, D.C., & Sha, F. (2020). Few-shot learning via embedding adaptation with set-to-set functions. In CVPR, pp. 8808\u20138817.","DOI":"10.1109\/CVPR42600.2020.00883"},{"key":"1917_CR72","doi-asserted-by":"crossref","unstructured":"Ye, H. J., Hu, H., & Zhan, D. C. (2021). Learning adaptive classifiers synthesis for generalized few-shot learning. IJCV, 129, 1930\u20131953.","DOI":"10.1007\/s11263-020-01381-4"},{"key":"1917_CR73","unstructured":"Yoon, S.W., Seo, J., & Moon, J. (2019). Tapnet: Neural network augmented with task-adaptive projection for few-shot learning. In ICML, PMLR, pp. 7115\u20137123."},{"key":"1917_CR74","doi-asserted-by":"crossref","unstructured":"Zhai, X., Wang, X., Mustafa, B., Steiner, A., Keysers, D., Kolesnikov, A., & Beyer, L. (2022). Lit: Zero-shot transfer with locked-image text tuning. In CVPR, pp. 18123\u201318133.","DOI":"10.1109\/CVPR52688.2022.01759"},{"key":"1917_CR75","doi-asserted-by":"crossref","unstructured":"Zhang, H., Zhang, L., Qi, X., Li, H., Torr, P.H., Koniusz, P. (2020). Few-shot action recognition with permutation-invariant attention. In ECCV, Springer, pp. 525\u2013542.","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"1917_CR76","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L., & Shum, H. (2022a). Dino: Detr with improved denoising anchor boxes for end-to-end object detection. In ICLR."},{"key":"1917_CR77","unstructured":"Zhang, R., Che, T., Ghahramani, Z., Bengio, Y., & Song, Y. (2018). Metagan: An adversarial approach to few-shot learning. NeurIPS 31."},{"key":"1917_CR78","unstructured":"Zhang, R., Fang, R., Gao, P., Zhang, W., Li, K., Dai, J., Qiao, Y., & Li, H. (2022b). Tip-adapter: Training-free clip-adapter for better vision-language modeling. In ECCV."},{"key":"1917_CR79","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhou, J., & He, X. (2021). Learning implicit temporal alignment for few-shot video classification. In IJCAI.","DOI":"10.24963\/ijcai.2021\/181"},{"key":"1917_CR80","doi-asserted-by":"crossref","unstructured":"Zheng, S., Chen, S., & Jin, Q. (2022). Few-shot action recognition with hierarchical matching and contrastive learning. In ECCV, Springer.","DOI":"10.1007\/978-3-031-19772-7_18"},{"key":"1917_CR81","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Yang, J., Zhang, P., Li, C., Codella, N., Li, L.H., Zhou, L., Dai, X., Yuan, L., & Li, Y., et\u00a0al. (2022). Regionclip: Region-based language-image pretraining. In CVPR, pp. 16793\u201316803.","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"1917_CR82","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., & Torralba, A. (2018). Temporal relational reasoning in videos. In ECCV, pp. 803\u2013818.","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"1917_CR83","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022a). Conditional prompt learning for vision-language models. In CVPR, pp. 16816\u201316825.","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"1917_CR84","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. IJCV, 130(9), 2337\u20132348.","journal-title":"IJCV"},{"key":"1917_CR85","doi-asserted-by":"crossref","unstructured":"Zhu, L., & Yang, Y. (2018). Compound memory networks for few-shot video classification. In ECCV, pp. 751\u2013766.","DOI":"10.1007\/978-3-030-01234-2_46"},{"issue":"1","key":"1917_CR86","first-page":"273","volume":"44","author":"L Zhu","year":"2020","unstructured":"Zhu, L., & Yang, Y. (2020). Label independent memory for semi-supervised few-shot video classification. IEEE Transactions on Pattern Analysis and Machine Intelligence., 44(1), 273\u201385.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence."},{"key":"1917_CR87","unstructured":"Zhu, X., Toisoul, A., Perez-Rua, J.M., Zhang, L., Martinez, B., & Xiang, T. (2021). Few-shot action recognition with prototype-centered attentive learning. arXiv preprint arXiv:2101.08085"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01917-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-023-01917-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01917-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,31]],"date-time":"2024-05-31T06:06:49Z","timestamp":1717135609000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-023-01917-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,17]]},"references-count":87,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["1917"],"URL":"https:\/\/doi.org\/10.1007\/s11263-023-01917-4","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,17]]},"assertion":[{"value":"1 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 September 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 October 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}