{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:56:01Z","timestamp":1783439761245,"version":"3.54.6"},"reference-count":60,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,8,23]],"date-time":"2023-08-23T00:00:00Z","timestamp":1692748800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,23]],"date-time":"2023-08-23T00:00:00Z","timestamp":1692748800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s13735-023-00286-5","type":"journal-article","created":{"date-parts":[[2023,8,23]],"date-time":"2023-08-23T15:01:51Z","timestamp":1692802911000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["CoCoOpter: Pre-train, prompt, and fine-tune the vision-language model for few-shot image classification"],"prefix":"10.1007","volume":"12","author":[{"given":"Jie","family":"Yan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxiang","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanming","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingmei","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoping","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xidao","family":"Luan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,8,23]]},"reference":[{"key":"286_CR1","doi-asserted-by":"crossref","unstructured":"Azizi S et al (2021) Big self-supervised models advance medical image classification. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3478\u20133488","DOI":"10.1109\/ICCV48922.2021.00346"},{"key":"286_CR2","doi-asserted-by":"crossref","unstructured":"Chen CFR, Fan Q, Panda R (2021) Crossvit: cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 357\u2013366","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"286_CR3","unstructured":"Dosovitskiy A et al (2021) An image is worth 16x16 words: transformers for image recognition at scale. In: International conference on learning representations"},{"issue":"6","key":"286_CR4","doi-asserted-by":"publisher","first-page":"2231","DOI":"10.1109\/TCSVT.2020.3016863","volume":"31","author":"Y Pei","year":"2021","unstructured":"Pei Y, Huang Y, Zhang X (2021) Consistency guided network for degraded image classification. IEEE Trans Circuits Syst Video Technol 31(6):2231\u20132246","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"286_CR5","doi-asserted-by":"crossref","unstructured":"Dai X et al (2021) Dynamic head: unifying object detection heads with attentions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7373\u20137382","DOI":"10.1109\/CVPR46437.2021.00729"},{"key":"286_CR6","doi-asserted-by":"crossref","unstructured":"Xie X, Cheng G, Wang J, Yao X, Han J (2021) Oriented r-cnn for object detection. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3520\u20133529","DOI":"10.1109\/ICCV48922.2021.00350"},{"key":"286_CR7","doi-asserted-by":"publisher","first-page":"4183","DOI":"10.1109\/TMM.2021.3114541","volume":"24","author":"C Yin","year":"2022","unstructured":"Yin C, Tang J, Yuan T, Xu Z, Wang Y (2022) Bridging the gap between semantic segmentation and instance segmentation. IEEE Trans Multimed 24:4183\u20134196. https:\/\/doi.org\/10.1109\/TMM.2021.3114541","journal-title":"IEEE Trans Multimed"},{"key":"286_CR8","doi-asserted-by":"publisher","first-page":"1035","DOI":"10.1109\/TMM.2020.2991592","volume":"23","author":"L Zhou","year":"2021","unstructured":"Zhou L, Gong C, Liu Z, Fu K (2021) SAL: selection and attention losses for weakly supervised semantic segmentation. IEEE Trans Multimed 23:1035\u20131048. https:\/\/doi.org\/10.1109\/TMM.2020.2991592","journal-title":"IEEE Trans Multimed"},{"issue":"6266","key":"286_CR9","doi-asserted-by":"publisher","first-page":"1332","DOI":"10.1126\/science.aab3050","volume":"350","author":"BM Lake","year":"2015","unstructured":"Lake BM, Salakhutdinov R, Tenenbaum JB (2015) Human-level concept learning through probabilistic program induction. Science 350(6266):1332\u20131338","journal-title":"Science"},{"key":"286_CR10","doi-asserted-by":"crossref","unstructured":"Deng J et al (2009) Imagenet: a large-scale hierarchical image database. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"286_CR11","doi-asserted-by":"crossref","unstructured":"Lin TY et al (2014) Microsoft coco: common objects in context. In: European conference on computer vision, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"286_CR12","doi-asserted-by":"crossref","unstructured":"Lifchitz Y, Avrithis Y, Picard S, Bursuc A (2019) Dense classification and implanting for few-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9258\u20139267","DOI":"10.1109\/CVPR.2019.00948"},{"key":"286_CR13","doi-asserted-by":"crossref","unstructured":"Liu Y, Schiele B, Sun Q (2020) An ensemble of epoch-wise empirical bayes for few-shot learning. In: European conference on computer vision, pp 404\u2013421","DOI":"10.1007\/978-3-030-58517-4_24"},{"key":"286_CR14","doi-asserted-by":"publisher","first-page":"9245","DOI":"10.1109\/TIP.2021.3124322","volume":"30","author":"C-C Lin","year":"2021","unstructured":"Lin C-C, Chu H-L, Wang Y-CF, Lei C-L (2021) Joint feature disentanglement and hallucination for few-shot image classification. IEEE Trans Image Process 30:9245\u20139258. https:\/\/doi.org\/10.1109\/TIP.2021.3124322","journal-title":"IEEE Trans Image Process"},{"key":"286_CR15","unstructured":"Chen W-Y, Liu Y-C, Kira Z, Wang Y-CF, Huang JB (2019) A closer look at few-shot classification. In: Proceedings of the international conference on learning representations, pp 1\u201324"},{"key":"286_CR16","doi-asserted-by":"crossref","unstructured":"Tian Y, Wang Y, Krishnan D, Tenenbaum JB, Isola P (2020) Rethinking few-shot image classification: a good embedding is all you need. In: European conference on computer vision, pp 266\u2013282","DOI":"10.1007\/978-3-030-58568-6_16"},{"key":"286_CR17","unstructured":"Finn C, Abbeel P, Levine S (2017) Model-agnostic meta-learning for fast adaptation of deep networks. In: Proceedings of the international conference on machine learning, pp 1126\u20131135"},{"key":"286_CR18","doi-asserted-by":"crossref","unstructured":"Sung F, Yang Y, Zhang L, Xiang T, Torr PHS, Hospedales TM (2018) Learning to compare: relation network for few-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1199\u20131208","DOI":"10.1109\/CVPR.2018.00131"},{"key":"286_CR19","unstructured":"Bertinetto L, Henriques JF, Valmadre J, Torr P, Vedaldi A (2016) Learning feed-forward one-shot learners. In: Proceedings of the advances in neural information processing systems, pp 523\u2013531"},{"key":"286_CR20","doi-asserted-by":"crossref","unstructured":"Chen Z, Fu Y, Wang Y-X, Ma L, Liu W, Hebert M (2019) Image deformation meta-networks for one-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8672\u20138681","DOI":"10.1109\/CVPR.2019.00888"},{"key":"286_CR21","doi-asserted-by":"crossref","unstructured":"Chen M et al (2020) Diversity transfer network for few-shot learning. In: Proceedings of the AAAI conference on artificial intelligence, pp 10559\u201310566","DOI":"10.1609\/aaai.v34i07.6628"},{"key":"286_CR22","doi-asserted-by":"publisher","unstructured":"Lin C-C, Wang Y-CF, Lei C-L, Chen K-T (2019) Semantics-guided data hallucination for few-shot visual classification. In: IEEE international conference on image processing (ICIP), pp 3302-3306. https:\/\/doi.org\/10.1109\/ICIP.2019.8803420","DOI":"10.1109\/ICIP.2019.8803420"},{"key":"286_CR23","doi-asserted-by":"crossref","unstructured":"Qi H, Brown M, Lowe DG (2018) Low-shot learning with imprinted weights. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 2018, pp 5822\u20135830","DOI":"10.1109\/CVPR.2018.00610"},{"key":"286_CR24","doi-asserted-by":"publisher","first-page":"1318","DOI":"10.1109\/TIP.2020.3043128","volume":"30","author":"X Li","year":"2021","unstructured":"Li X, Wu J, Sun Z, Ma Z, Cao J, Xue J-H (2021) BSNet: Bi-similarity network for few-shot fine-grained image classification. IEEE Trans Image Process 30:1318\u20131331. https:\/\/doi.org\/10.1109\/TIP.2020.3043128","journal-title":"IEEE Trans Image Process"},{"key":"286_CR25","doi-asserted-by":"publisher","first-page":"1200","DOI":"10.1109\/TMM.2020.2993952","volume":"23","author":"Y Zhu","year":"2021","unstructured":"Zhu Y, Min W, Jiang S (2021) Attribute-guided feature learning for few-shot image recognition. IEEE Trans Multimed 23:1200\u20131209. https:\/\/doi.org\/10.1109\/TMM.2020.2993952","journal-title":"IEEE Trans Multimed"},{"key":"286_CR26","unstructured":"Radford A et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763"},{"key":"286_CR27","unstructured":"Jia C et al (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, pp 4904\u20134916"},{"issue":"9","key":"286_CR28","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Learning to prompt for vision language models. Int J Comput Vis 130(9):2337\u20132348","journal-title":"Int J Comput Vis"},{"issue":"10","key":"286_CR29","doi-asserted-by":"publisher","first-page":"1872","DOI":"10.1007\/s11431-020-1647-3","volume":"63","author":"X Qiu","year":"2020","unstructured":"Qiu X et al (2020) Pre-trained models for natural language processing: a survey. Sci China Technol Sci 63(10):1872\u20131897","journal-title":"Sci China Technol Sci"},{"key":"286_CR30","doi-asserted-by":"crossref","unstructured":"Li M et al (2022) Bridge-prompt: towards ordinal action understanding in instructional videos. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19880\u201319889","DOI":"10.1109\/CVPR52688.2022.01926"},{"key":"286_CR31","doi-asserted-by":"crossref","unstructured":"Zhou K, Yang J, Loy C C, Liu Z (2022) Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16816\u201316825","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"286_CR32","doi-asserted-by":"crossref","unstructured":"Zeng Y et al (2022) Point prompt tuning for temporally language grounding. In: Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval, pp 2003\u20132007","DOI":"10.1145\/3477495.3531795"},{"key":"286_CR33","doi-asserted-by":"crossref","unstructured":"Rao Y et al (2022) DenseCLIP: language-guided dense prediction with context-aware prompting. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18082\u201318091","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"286_CR34","doi-asserted-by":"crossref","unstructured":"Chen X et al (2022) Knowprompt: knowledge-aware prompt-tuning with synergistic optimization for relation extraction. In Proceedings of the ACM web conference, pp 2778\u20132788","DOI":"10.1145\/3485447.3511998"},{"key":"286_CR35","unstructured":"Gao P et al (2021) Clip-adapter: better vision-language models with feature adapters. arXiv:2110.04544"},{"issue":"4","key":"286_CR36","doi-asserted-by":"publisher","first-page":"594","DOI":"10.1109\/TPAMI.2006.79","volume":"28","author":"FF Li","year":"2006","unstructured":"Li FF, Fergus R, Perona P (2006) One-shot learning of object categories. IEEE Trans Pattern Anal Mach Intell 28(4):594\u2013611. https:\/\/doi.org\/10.1109\/TPAMI.2006.79","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"286_CR37","unstructured":"Lee Y, Choi S (2018) Gradient-based meta-learning with learned layerwise metric and subspace. In: International conference on machine learning, PMLR, pp 2927\u20132936"},{"key":"286_CR38","unstructured":"Ravi S, Larochelle H (2017) Optimization as a model for few-shot learning. In: International conference on learning representations"},{"key":"286_CR39","doi-asserted-by":"crossref","unstructured":"Li W, Wang L, Xu J, Huo J, Gao Y, Luo J (2019) Revisiting local descriptor based image-to-class measure for few-shot learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7260\u20137268","DOI":"10.1109\/CVPR.2019.00743"},{"key":"286_CR40","doi-asserted-by":"crossref","unstructured":"Zhang H, Zhang J, Koniusz P (2019) Few-shot learning via saliency-guided hallucination of samples. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2770\u20132779","DOI":"10.1109\/CVPR.2019.00288"},{"key":"286_CR41","doi-asserted-by":"crossref","unstructured":"Wang W, Bao H, Dong L, et al (2022) Image as a foreign language: BEiT pretraining for all vision and vision-language tasks. arXiv:2208.10442","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"286_CR42","unstructured":"Vaswani A et al (2017) Attention is all you need. In: Advances in neural information processing systems, vol 30"},{"key":"286_CR43","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"286_CR44","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805"},{"key":"286_CR45","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I (2018) Improving language understanding by generative pre-training. https:\/\/s3-us-west-2.amazonaws.com\/openai-assets\/research-covers\/language-unsupervised\/language_understanding_paper.pdf"},{"key":"286_CR46","unstructured":"Zhao Z, Wallace E, Feng S, Klein D, Singh S (2021) Calibrate before use: improving few-shot performance of language models. In: International conference on machine learning, pp 12697\u201312706"},{"key":"286_CR47","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1162\/tacl_a_00324","volume":"8","author":"Z Jiang","year":"2020","unstructured":"Jiang Z, Xu FF, Araki J, Neubig G (2020) How can we know what language models know? Trans Assoc Comput Linguist 8:423\u2013438","journal-title":"Trans Assoc Comput Linguist"},{"key":"286_CR48","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems, vol 25"},{"key":"286_CR49","doi-asserted-by":"crossref","unstructured":"Zhu JY et al (2017) Unpaired image-to-image translation using cycle-consistent adversarial networks. In: Proceedings of the IEEE international conference on computer vision, pp 2223\u20132232","DOI":"10.1109\/ICCV.2017.244"},{"key":"286_CR50","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1162\/tacl_a_00343","volume":"8","author":"Y Liu","year":"2020","unstructured":"Liu Y et al (2020) Multilingual denoising pre-training for neural machine translation. Trans Assoc Comput Linguist 8:726\u2013742","journal-title":"Trans Assoc Comput Linguist"},{"key":"286_CR51","doi-asserted-by":"crossref","unstructured":"Parkhi OM, Vedaldi A, Zisserman A, Jawahar C (2012) Cats and dogs. In: Proceedings of the IEEE\/CVF international conference on computer vision and pattern recognition, pp 3498\u20133505","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"286_CR52","doi-asserted-by":"crossref","unstructured":"Krause J, Stark M, Deng J, Fei-Fei L (2013) 3d object representations for fine-grained categorization. In: Proceedings of the IEEE\/CVF international conference on computer vision and pattern recognition, pp 554\u2013561","DOI":"10.1109\/ICCVW.2013.77"},{"key":"286_CR53","doi-asserted-by":"crossref","unstructured":"Nilsback ME, Zisserman A (2008) Automated flower classification over a large number of classes. In: 2008 6th Indian conference on computer vision, graphics & image processing, pp 722\u2013729","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"286_CR54","doi-asserted-by":"crossref","unstructured":"Bossard L, Guillaumin M, Gool LV (2014) Food-101-mining discriminative components with random forests. In: European conference on computer vision, pp 446\u2013461","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"286_CR55","unstructured":"Maji S, Rahtu E, Kannala J, Blaschko M, Vedaldi A (2013) Fine-grained visual classification of aircraft. arXiv:1306.5151"},{"key":"286_CR56","doi-asserted-by":"crossref","unstructured":"Xiao J, Hays J, Ehinger KA, Oliva A, Torralba A (2010) Sun database: large-scale scene recognition from abbey to zoo. In: Proceedings of the IEEE\/CVF international conference on computer vision and pattern recognition, pp 3485\u20133492","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"286_CR57","unstructured":"Soomro K, Zamir AR, Shah M (2012) UCF101: a dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402"},{"key":"286_CR58","doi-asserted-by":"crossref","unstructured":"Cimpoi M, Maji S, Kokkinos I, Mohamed S, Vedaldi A (2014) Describing textures in the wild. In: Proceedings of the IEEE\/CVF international conference on computer vision and pattern recognition, pp 3606\u20133613","DOI":"10.1109\/CVPR.2014.461"},{"issue":"7","key":"286_CR59","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber P, Bischke B, Dengel A, Borth D (2019) EuroSAT: a novel dataset and deep learning benchmark for land use and land cover classification. IEEE J Sel Top Appl Earth Obs Remote Sens 12(7):2217\u20132226","journal-title":"IEEE J Sel Top Appl Earth Obs Remote Sens"},{"key":"286_CR60","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton G (2020) A simple framework for contrastive learning of visual representations. In: International conference on machine learning, pp 1597\u20131607"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00286-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-023-00286-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00286-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,2]],"date-time":"2023-12-02T14:13:06Z","timestamp":1701526386000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-023-00286-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,23]]},"references-count":60,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["286"],"URL":"https:\/\/doi.org\/10.1007\/s13735-023-00286-5","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,23]]},"assertion":[{"value":"21 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 May 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 July 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 August 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with Ethical Standards"}}],"article-number":"27"}}