{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:27:17Z","timestamp":1783438037495,"version":"3.54.6"},"reference-count":106,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T00:00:00Z","timestamp":1724716800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T00:00:00Z","timestamp":1724716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372155"],"award-info":[{"award-number":["62372155"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Joint Fund of Ministry of Education for Equipment Pre-research","award":["8091B022123"],"award-info":[{"award-number":["8091B022123"]}]},{"name":"Research Fund from Science and Technology on Underwater Vehicle Technology Laboratory","award":["2021JCJQ-SYSJJ-LB06905"],"award-info":[{"award-number":["2021JCJQ-SYSJJ-LB06905"]}]},{"name":"Key Laboratory of Information System Requirements","award":["LHZZ 2021-M04"],"award-info":[{"award-number":["LHZZ 2021-M04"]}]},{"name":"Water Science and Technology Project of Jiangsu Province","award":["2021063"],"award-info":[{"award-number":["2021063"]}]},{"name":"Qinglan Project of Jiangsu Province"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-024-10915-y","type":"journal-article","created":{"date-parts":[[2024,8,27]],"date-time":"2024-08-27T01:02:29Z","timestamp":1724720549000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":51,"title":["Few-shot adaptation of multi-modal foundation models: a survey"],"prefix":"10.1007","volume":"57","author":[{"given":"Fan","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianshu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenwen","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuanyi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenwen","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaocong","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Delong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,8,27]]},"reference":[{"key":"10915_CR1","unstructured":"Baevski A, Hsu W, Xu Q, Babu A, Gu J, Auli M (2022) data2vec: A general framework for self-supervised learning in speech, vision and language. In: Chaudhuri, K., Jegelka, S., Song, L., Szepesv\u00e1ri, C., Niu, G., Sabato, S. (eds.) International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA. Proceedings of Machine Learning Research, vol. 162, pp. 1298\u20131312. PMLR, ???. https:\/\/proceedings.mlr.press\/v162\/baevski22a.html"},{"key":"10915_CR2","unstructured":"Bahng H, Jahanian A, Sankaranarayanan S, Isola P (2022) Exploring Visual Prompts for Adapting Large-Scale Models"},{"key":"10915_CR3","unstructured":"Bao H, Dong L, Piao S, Wei F (2022) Beit: BERT pre-training of image transformers. In: The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022. OpenReview.net, ???. https:\/\/openreview.net\/forum?id=p-BhZSz59o4"},{"key":"10915_CR4","doi-asserted-by":"publisher","unstructured":"Bossard L, Guillaumin M, Gool LV (2014) Food-101 - mining discriminative components with random forests. In: Fleet, D.J., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision - ECCV 2014 - 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part VI. Lecture Notes in Computer Science, vol. 8694, pp. 446\u2013461. Springer, ???. https:\/\/doi.org\/10.1007\/978-3-319-10599-4_29","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"10915_CR5","doi-asserted-by":"publisher","unstructured":"Bulat A, Tzimiropoulos G (2023) LASP: text-to-text optimization for language-aware soft prompting of vision & language models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, Vancouver, BC, Canada, June 17-24, 2023, pp. 23232\u201323241. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52729.2023.02225","DOI":"10.1109\/CVPR52729.2023.02225"},{"key":"10915_CR6","unstructured":"Cai H, Gan C, Wang T, Zhang Z, Han S (2020) Once-for-all: Train one network and specialize it for efficient deployment. In: 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020. OpenReview.net, ??? . https:\/\/openreview.net\/forum?id=HylxE1HKwS"},{"key":"10915_CR7","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton GE (2020) A simple framework for contrastive learning of visual representations. In: Proceedings of the 37th International Conference on Machine Learning, ICML 2020, 13-18 July 2020, Virtual Event. Proceedings of Machine Learning Research, vol. 119, pp. 1597\u20131607. PMLR, ???. http:\/\/proceedings.mlr.press\/v119\/chen20j.html"},{"key":"10915_CR8","doi-asserted-by":"publisher","unstructured":"Chen D, Wu Z, Liu F, Yang Z, Huang Y, Bao Y, Zhou E (2022) Prototypical contrastive language image pretraining. CoRR https:\/\/doi.org\/10.48550\/arXiv.2206.10996arXiv:2206.10996","DOI":"10.48550\/arXiv.2206.10996"},{"key":"10915_CR9","unstructured":"Chen G, Yao W, Song X, Li X, Rao Y, Zhang K (2023) PLOT: prompt learning with optimal transport for vision-language models. In: The Eleventh International Conference on Learning Representations, ICLR 2023, Kigali, Rwanda, May 1-5, 2023. OpenReview.net, ???. https:\/\/openreview.net\/pdf?id=zqwryBoXYnh"},{"key":"10915_CR10","doi-asserted-by":"publisher","unstructured":"Chowdhery A, Narang S, Devlin J, Bosma M, Mishra G, Roberts A, Barham P, Chung HW, Sutton C, Gehrmann S, Schuh P, Shi K, Tsvyashchenko S, Maynez J, Rao A, Barnes P, Tay Y, Shazeer N, Prabhakaran V, Reif E, Du N, Hutchinson B, Pope R, Bradbury J, Austin J, Isard M, Gur-Ari G, Yin P, Duke T, Levskaya A, Ghemawat S, Dev S, Michalewski H, Garcia X, Misra V, Robinson K, Fedus L, Zhou D, Ippolito D, Luan D, Lim H, Zoph B, Spiridonov A, Sepassi R, Dohan D, Agrawal S, Omernick M, Dai AM, Pillai TS, Pellat M, Lewkowycz A, Moreira E, Child R, Polozov O, Lee K, Zhou Z, Wang X, Saeta B, Diaz M, Firat O, Catasta M, Wei J, Meier-Hellstern K, Eck D, Dean J, Petrov S, Fiedel N (2022) Palm: Scaling language modeling with pathways. CoRR https:\/\/doi.org\/10.48550\/arXiv.2204.02311arXiv:abs\/2204.02311","DOI":"10.48550\/arXiv.2204.02311"},{"key":"10915_CR11","doi-asserted-by":"publisher","unstructured":"Cimpoi M, Maji S, Kokkinos I, Mohamed S, Vedaldi A (2014) Describing textures in the wild. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2014, Columbus, OH, USA, June 23-28, 2014, pp. 3606\u20133613. IEEE Computer Society, ??? . https:\/\/doi.org\/10.1109\/CVPR.2014.461","DOI":"10.1109\/CVPR.2014.461"},{"key":"10915_CR12","doi-asserted-by":"crossref","unstructured":"Crammer K, Kearns MJ, Wortman J (2006) Learning from multiple sources. In: Sch\u00f6lkopf, B., Platt, J.C., Hofmann, T. (eds.) Advances in Neural Information Processing Systems 19, Proceedings of the Twentieth Annual Conference on Neural Information Processing Systems, Vancouver, British Columbia, Canada, December 4-7, 2006, pp. 321\u2013328. MIT Press, ???. https:\/\/proceedings.neurips.cc\/paper\/2006\/hash\/0f21f0349462cacdc5796990d37760ae-Abstract.html","DOI":"10.7551\/mitpress\/7503.003.0045"},{"key":"10915_CR13","doi-asserted-by":"publisher","unstructured":"Deng J, Dong W, Socher R, Li L, Li K, Fei-Fei L (2009) Imagenet: A large-scale hierarchical image database. In: 2009 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2009), 20-25 June 2009, Miami, Florida, USA, pp. 248\u2013255. IEEE Computer Society, ???. https:\/\/doi.org\/10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"10915_CR14","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: Pre-training of deep bidirectional transformers for language understanding. In: Burstein, J., Doran, C., Solorio, T. (eds.) Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics, Minneapolis, Minnesota. https:\/\/doi.org\/10.18653\/v1\/N19-1423 . https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"10915_CR15","doi-asserted-by":"publisher","unstructured":"Ding N, Qin Y, Yang G, Wei F, Yang Z, Su Y, Hu S, Chen Y, Chan C, Chen W, Yi J, Zhao W, Wang X, Liu Z, Zheng H, Chen J, Liu Y, Tang J, Li J, Sun M (2022) Delta tuning: A comprehensive study of parameter efficient methods for pre-trained language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2203.06904arXiv:2203.06904","DOI":"10.48550\/arXiv.2203.06904"},{"key":"10915_CR16","doi-asserted-by":"publisher","unstructured":"Ding K, Wang Y, Liu P, Yu Q, Zhang H, Xiang S, Pan C (2022) Prompt tuning with soft context sharing for vision-language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2208.13474arXiv:2208.13474","DOI":"10.48550\/arXiv.2208.13474"},{"issue":"1","key":"10915_CR17","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1016\/j.cviu.2005.09.012","volume":"106","author":"L Fei-Fei","year":"2007","unstructured":"Fei-Fei L, Fergus R, Perona P (2007) Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. Comput Vis Image Underst 106(1):59\u201370. https:\/\/doi.org\/10.1016\/j.cviu.2005.09.012","journal-title":"Comput Vis Image Underst"},{"key":"10915_CR18","unstructured":"F\u00fcrst A, Rumetshofer E, Lehner J, Tran VT, Tang F, Ramsauer H, Kreil DP, Kopp M, Klambauer G, Bitto A, Hochreiter S (2022) CLOOB: modern hopfield networks with infoloob outperform CLIP. In: NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/8078e76f913e31b8467e85b4c0f0d22b-Abstract-Conference.html"},{"key":"10915_CR19","unstructured":"Gao P, Geng S, Zhang R, Ma T, Fang R, Zhang Y, Li H, Qiao Y (2021) Clip-adapter: Better vision-language models with feature adapters. CoRR arXiv:2110.04544"},{"key":"10915_CR20","unstructured":"Gao Y, Liu J, Xu Z, Zhang J, Li K, Ji R, Shen C (2022) Pyramidclip: Hierarchical feature alignment for vision-language model pretraining. In: NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/e9882f7f7c44a10acc01132302bac9d8-Abstract-Conference.html"},{"key":"10915_CR21","doi-asserted-by":"crossref","unstructured":"Guo Z, Zhang R, Qiu L, Ma X, Miao X, He X, Cui B (2023) CALIP: zero-shot enhancement of CLIP with parameter-free attention. In: Williams, B., Chen, Y., Neville, J. (eds.) Thirty-Seventh AAAI Conference on Artificial Intelligence, AAAI 2023, Thirty-Fifth Conference on Innovative Applications of Artificial Intelligence, IAAI 2023, Thirteenth Symposium on Educational Advances in Artificial Intelligence, EAAI 2023, Washington, DC, USA, February 7-14, 2023, pp. 746\u2013754. AAAI Press, ???. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/25152","DOI":"10.1609\/aaai.v37i1.25152"},{"key":"10915_CR22","doi-asserted-by":"publisher","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick RB (2022) Masked autoencoders are scalable vision learners. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 15979\u201315988. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52688.2022.01553","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10915_CR23","doi-asserted-by":"publisher","unstructured":"He K, Fan H, Wu Y, Xie S, Girshick RB (2020) Momentum contrast for unsupervised visual representation learning. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2020, Seattle, WA, USA, June 13-19, 2020, pp. 9726\u20139735. Computer Vision Foundation \/ IEEE, ???. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00975","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"10915_CR24","doi-asserted-by":"publisher","unstructured":"Helber P, Bischke B, Dengel A, Borth D (2019) Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE J. Sel. Top. Appl. Earth Obs. Remote. Sens. 12(7), 2217\u20132226 https:\/\/doi.org\/10.1109\/JSTARS.2019.2918242","DOI":"10.1109\/JSTARS.2019.2918242"},{"key":"10915_CR25","doi-asserted-by":"publisher","unstructured":"Hendrycks D, Basart S, Mu N, Kadavath S, Wang F, Dorundo E, Desai R, Zhu T, Parajuli S, Guo M, Song D, Steinhardt J, Gilmer J (2021) The many faces of robustness: A critical analysis of out-of-distribution generalization. In: 2021 IEEE\/CVF International Conference on Computer Vision, ICCV 2021, Montreal, QC, Canada, October 10-17, 2021, pp. 8320\u20138329. IEEE, ??? . https:\/\/doi.org\/10.1109\/ICCV48922.2021.00823","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"10915_CR26","doi-asserted-by":"publisher","unstructured":"Hendrycks D, Zhao K, Basart S, Steinhardt J, Song D (2021) Natural adversarial examples. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15257\u201315266. https:\/\/doi.org\/10.1109\/CVPR46437.2021.01501","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"10915_CR27","unstructured":"Houlsby N, Giurgiu A, Jastrzebski S, Morrone B, Laroussilhe Q, Gesmundo A, Attariyan M, Gelly S (2019) Parameter-efficient transfer learning for NLP. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9-15 June 2019, Long Beach, California, USA. Proceedings of Machine Learning Research, vol. 97, pp. 2790\u20132799. PMLR, ???. http:\/\/proceedings.mlr.press\/v97\/houlsby19a.html"},{"key":"10915_CR28","doi-asserted-by":"publisher","unstructured":"Huang T, Chu J, Wei F (2022) Unsupervised prompt learning for vision-language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2204.03649arXiv:2204.03649","DOI":"10.48550\/arXiv.2204.03649"},{"key":"10915_CR29","unstructured":"Huo Y, Zhang M, Liu G, Lu H, Gao Y, Yang G, Wen J, Zhang H, Xu B, Zheng W, Xi Z, Yang Y, Hu A, Zhao J, Li R, Zhao Y, Zhang L, Song Y, Hong X, Cui W, Hou DY, Li Y, Li J, Liu P, Gong Z, Jin C, Sun Y, Chen S, Lu Z, Dou Z, Jin Q, Lan Y, Zhao WX, Song R, Wen J(2021) Wenlan: Bridging vision and language by large-scale multi-modal pre-training. CoRR arXiv:2103.06561"},{"key":"10915_CR30","doi-asserted-by":"publisher","unstructured":"Jiang H, Zhang J, Huang R, Ge C, Ni Z, Lu J, Zhou J, Song S, Huang G (2022) Cross-modal adapter for text-video retrieval. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.09623arXiv:2211.09623","DOI":"10.48550\/arXiv.2211.09623"},{"key":"10915_CR31","doi-asserted-by":"publisher","unstructured":"Jia M, Tang L, Chen B, Cardie C, Belongie SJ, Hariharan B, Lim S (2022) Visual prompt tuning. In: Avidan, S., Brostow, G.J., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXXIII. Lecture Notes in Computer Science, vol. 13693, pp. 709\u2013727. Springer, ???. https:\/\/doi.org\/10.1007\/978-3-031-19827-4_41","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"10915_CR32","unstructured":"Jia C, Yang Y, Xia Y, Chen Y, Parekh Z, Pham H, Le QV, Sung Y, Li Z, Duerig T (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol. 139, pp. 4904\u20134916. PMLR, ???. http:\/\/proceedings.mlr.press\/v139\/jia21b.html"},{"key":"10915_CR33","doi-asserted-by":"publisher","unstructured":"Jie S, Deng Z (2022) Convolutional bypasses are better vision transformer adapters. CoRR https:\/\/doi.org\/10.48550\/arXiv.2207.07039arXiv:2207.07039","DOI":"10.48550\/arXiv.2207.07039"},{"key":"10915_CR34","doi-asserted-by":"crossref","unstructured":"Jie S, Deng Z (2023) Fact: Factor-tuning for lightweight adaptation on vision transformer. In: Williams, B., Chen, Y., Neville, J. (eds.) Thirty-Seventh AAAI Conference on Artificial Intelligence, AAAI 2023, Thirty-Fifth Conference on Innovative Applications of Artificial Intelligence, IAAI 2023, Thirteenth Symposium on Educational Advances in Artificial Intelligence, EAAI 2023, Washington, DC, USA, February 7-14, 2023, pp. 1060\u20131068. AAAI Press, ??? . https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/25187","DOI":"10.1609\/aaai.v37i1.25187"},{"issue":"3","key":"10915_CR35","doi-asserted-by":"publisher","first-page":"1713","DOI":"10.3390\/s23031713","volume":"23","author":"J Kang","year":"2023","unstructured":"Kang J, Kang JK, Kim J, Jeon K, Chung H, Park B (2023) Neural architecture search survey: A computer vision perspective. Sensors 23(3):1713. https:\/\/doi.org\/10.3390\/s23031713","journal-title":"Sensors"},{"key":"10915_CR36","doi-asserted-by":"publisher","unstructured":"Khattak MU, Rasheed HA, Maaz M, Khan S, Khan FS (2022) Maple: Multi-modal prompt learning. CoRR https:\/\/doi.org\/10.48550\/arXiv.2210.03117arXiv:2210.03117","DOI":"10.48550\/arXiv.2210.03117"},{"key":"10915_CR37","doi-asserted-by":"publisher","unstructured":"Krause J, Stark M, Deng J, Fei-Fei L (2013) 3d object representations for fine-grained categorization. In: 2013 IEEE International Conference on Computer Vision Workshops, ICCV Workshops 2013, Sydney, Australia, December 1-8, 2013, pp. 554\u2013561. IEEE Computer Society, ???. https:\/\/doi.org\/10.1109\/ICCVW.2013.77","DOI":"10.1109\/ICCVW.2013.77"},{"key":"10915_CR38","doi-asserted-by":"publisher","unstructured":"Lester B, Al-Rfou R, Constant N (2021) The power of scale for parameter-efficient prompt tuning. In: Moens, M., Huang, X., Specia, L., Yih, S.W. (eds.) Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP 2021, Virtual Event \/ Punta Cana, Dominican Republic, 7-11 November, 2021, pp. 3045\u20133059. Association for Computational Linguistics, ??? . https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.243","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"10915_CR39","doi-asserted-by":"publisher","unstructured":"Li XL, Liang P (2021) Prefix-tuning: Optimizing continuous prompts for generation. In: Zong, C., Xia, F., Li, W., Navigli, R. (eds.) Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP 2021, (Volume 1: Long Papers), Virtual Event, August 1-6, 2021, pp. 4582\u20134597. Association for Computational Linguistics, ??? . https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.353","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"10915_CR40","doi-asserted-by":"publisher","unstructured":"Li J, Gao M, Wei L, Tang S, Zhang W, Li M, Ji W, Tian Q, Chua T, Zhuang Y (2023) Gradient-regulated meta-prompt learning for generalizable vision-language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2303.06571arXiv:2303.06571","DOI":"10.48550\/arXiv.2303.06571"},{"key":"10915_CR41","unstructured":"Li Y, Liang F, Zhao L, Cui Y, Ouyang W, Shao J, Yu F, Yan J (2022) Supervision exists everywhere: A data efficient contrastive language-image pre-training paradigm. In: The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022. OpenReview.net, ???. https:\/\/openreview.net\/forum?id=zq1iJkNk3uN"},{"key":"10915_CR42","unstructured":"Li C, Liu H, Li LH, Zhang P, Aneja J, Yang J, Jin P, Hu H, Liu Z, Lee YJ, Gao J (2022) ELEVATER: A benchmark and toolkit for evaluating language-augmented visual models. In: NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/3c4688b6a76f25f2311daa0d75a58f1a-Abstract-Datasets_and_Benchmarks.html"},{"key":"10915_CR43","unstructured":"Li J, Li D, Xiong C, Hoi SCH (2022) BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: Chaudhuri, K., Jegelka, S., Song, L., Szepesv\u00e1ri, C., Niu, G., Sabato, S. (eds.) International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA. Proceedings of Machine Learning Research, vol. 162, pp. 12888\u201312900. PMLR, ??? . https:\/\/proceedings.mlr.press\/v162\/li22n.html"},{"key":"10915_CR44","doi-asserted-by":"publisher","unstructured":"Lin Z, Yu S, Kuang Z, Pathak D, Ramanan D (2023) Multimodality helps unimodality: Cross-modal few-shot learning with multimodal models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, Vancouver, BC, Canada, June 17-24, 2023, pp. 19325\u201319337. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52729.2023.01852","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"10915_CR45","unstructured":"Liu M, Li B, Yu Y (2024) Fully Fine-tuned CLIP Models are Efficient Few-Shot Learners . https:\/\/arxiv.org\/abs\/2407.04003"},{"key":"10915_CR46","doi-asserted-by":"publisher","unstructured":"Liu X, Wang D, Li M, Duan Z, Xu Y, Chen B, Zhou M (2023) Patch-token aligned bayesian prompt learning for vision-language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2303.09100arXiv:2303.09100","DOI":"10.48550\/arXiv.2303.09100"},{"key":"10915_CR47","unstructured":"Liu X, Zheng Y, Du Z, Ding M, Qian Y, Yang Z, Tang J (2021) GPT understands, too. CoRR arXiv:2103.10385"},{"key":"10915_CR48","doi-asserted-by":"crossref","unstructured":"Liu W, Zhou P, Zhao Z, Wang Z, Ju Q, Deng H, Wang P (2020) K-BERT: enabling language representation with knowledge graph. In: The Thirty-Fourth AAAI Conference on Artificial Intelligence, AAAI 2020, The Thirty-Second Innovative Applications of Artificial Intelligence Conference, IAAI 2020, The Tenth AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI 2020, New York, NY, USA, February 7-12, 2020, pp. 2901\u20132908. AAAI Press, ???. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/5681","DOI":"10.1609\/aaai.v34i03.5681"},{"key":"10915_CR49","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Wallach, H.M., Larochelle, H., Beygelzimer, A., d\u2019Alch\u00e9-Buc, F., Fox, E.B., Garnett, R. (eds.) Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, pp. 13\u201323 . https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/c74d97b01eae257e44aa9d5bade97baf-Abstract.html"},{"key":"10915_CR50","doi-asserted-by":"publisher","unstructured":"Lu H, Ding M, Huo Y, Yang G, Lu Z, Tomizuka M, Zhan W (2023) Uniadapter: Unified parameter-efficient transfer learning for cross-modal modeling. CoRR https:\/\/doi.org\/10.48550\/arXiv.2302.06605arXiv:2302.06605","DOI":"10.48550\/arXiv.2302.06605"},{"key":"10915_CR51","doi-asserted-by":"publisher","unstructured":"Lu Y, Liu J, Zhang Y, Liu Y, Tian X (2022) Prompt distribution learning. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 5196\u20135205. IEEE, ???. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00514","DOI":"10.1109\/CVPR52688.2022.00514"},{"issue":"9","key":"10915_CR52","doi-asserted-by":"publisher","first-page":"4616","DOI":"10.1109\/TCSVT.2023.3245584","volume":"33","author":"C Ma","year":"2023","unstructured":"Ma C, Liu Y, Deng J, Xie L, Dong W, Xu C (2023) Understanding and mitigating overfitting in prompt tuning for vision-language models. IEEE Trans Circuits Syst Video Technol 33(9):4616\u20134629. https:\/\/doi.org\/10.1109\/TCSVT.2023.3245584","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"10915_CR53","unstructured":"Maji S, Rahtu E, Kannala J, Blaschko MB, Vedaldi A (2013) Fine-grained visual classification of aircraft. CoRR arXiv:1306.5151"},{"key":"10915_CR54","unstructured":"Maurer A (2004) A note on the PAC bayesian theorem. CoRR cs.LG\/0411099"},{"key":"10915_CR55","unstructured":"Menon S, Vondrick C (2023) Visual classification via description from large language models. In: The Eleventh International Conference on Learning Representations, ICLR 2023, Kigali, Rwanda, May 1-5, 2023. OpenReview.net, ???. https:\/\/openreview.net\/pdf?id=jlAjNL8z5cs"},{"key":"10915_CR56","doi-asserted-by":"publisher","unstructured":"Mu N, Kirillov A, Wagner DA, Xie S (2022) SLIP: self-supervision meets language-image pre-training. In: Avidan, S., Brostow, G.J., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXVI. Lecture Notes in Computer Science, vol. 13686, pp. 529\u2013544. Springer, ??? . https:\/\/doi.org\/10.1007\/978-3-031-19809-0_30","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"10915_CR57","doi-asserted-by":"publisher","unstructured":"Nilsback M, Zisserman A (2008) Automated flower classification over a large number of classes. In: Sixth Indian Conference on Computer Vision, Graphics & Image Processing, ICVGIP 2008, Bhubaneswar, India, 16-19 December 2008, pp. 722\u2013729. IEEE Computer Society, ??? . https:\/\/doi.org\/10.1109\/ICVGIP.2008.47","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"10915_CR58","doi-asserted-by":"publisher","unstructured":"OpenAI (2023) GPT-4 technical report. CoRR https:\/\/doi.org\/10.48550\/arXiv.2303.08774arXiv:2303.08774","DOI":"10.48550\/arXiv.2303.08774"},{"key":"10915_CR59","unstructured":"Pantazis O, Brostow GJ, Jones KE, Aodha OM (2022) Svl-adapter: Self-supervised adapter for vision-language pretrained models. In: 33rd British Machine Vision Conference 2022, BMVC 2022, London, UK, November 21-24, 2022, p. 580. BMVA Press, ???. https:\/\/bmvc2022.mpi-inf.mpg.de\/580\/"},{"key":"10915_CR60","doi-asserted-by":"publisher","unstructured":"Parkhi OM, Vedaldi A, Zisserman A, Jawahar CV (2012) Cats and dogs. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, Providence, RI, USA, June 16-21, 2012, pp. 3498\u20133505. IEEE Computer Society, ??? . https:\/\/doi.org\/10.1109\/CVPR.2012.6248092","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"10915_CR61","doi-asserted-by":"publisher","unstructured":"Peng F, Yang X, Xu C (2023) Sgva-clip: Semantic-guided visual adapting of vision-language models for few-shot image classification. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.16191arXiv:2211.16191","DOI":"10.48550\/arXiv.2211.16191"},{"key":"10915_CR62","doi-asserted-by":"publisher","unstructured":"Pratt SM, Liu R, Farhadi A (2022) What does a platypus look like? generating customized prompts for zero-shot image classification. CoRR https:\/\/doi.org\/10.48550\/arXiv.2209.03320arXiv:2209.03320","DOI":"10.48550\/arXiv.2209.03320"},{"key":"10915_CR63","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, Krueger G, Sutskever I (2021) Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol. 139, pp. 8748\u20138763. PMLR, ???. http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"10915_CR64","unstructured":"Radford A, Narasimhan K (2018) Improving language understanding by generative pre-training. https:\/\/api.semanticscholar.org\/CorpusID:49313245"},{"key":"10915_CR65","unstructured":"Recht B, Roelofs R, Schmidt L, Shankar V (2019) Do imagenet classifiers generalize to imagenet? In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9-15 June 2019, Long Beach, California, USA. Proceedings of Machine Learning Research, vol. 97, pp. 5389\u20135400. PMLR, ???. http:\/\/proceedings.mlr.press\/v97\/recht19a.html"},{"key":"10915_CR66","doi-asserted-by":"publisher","unstructured":"Reynolds L, McDonell K (2021) Prompt programming for large language models: Beyond the few-shot paradigm. In: Kitamura, Y., Quigley, A., Isbister, K., Igarashi, T. (eds.) CHI \u201921: CHI Conference on Human Factors in Computing Systems, Virtual Event \/ Yokohama Japan, May 8-13, 2021, Extended Abstracts, pp. 314\u201313147. ACM, ??? . https:\/\/doi.org\/10.1145\/3411763.3451760","DOI":"10.1145\/3411763.3451760"},{"key":"10915_CR67","doi-asserted-by":"publisher","unstructured":"Rombach R, Blattmann A, Lorenz D, Esser P, Ommer B (2022) High-resolution image synthesis with latent diffusion models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 10674\u201310685. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52688.2022.01042","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"10915_CR68","unstructured":"Schuhmann C, Beaumont R, Vencu R, Gordon C, Wightman R, Cherti M, Coombes T, Katta A, Mullis C, Wortsman M, Schramowski P, Kundurthy S, Crowson K, Schmidt L, Kaczmarczyk R, Jitsev J (2022) LAION-5B: an open large-scale dataset for training next generation image-text models. In: NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/a1859debfb3b59d094f3504d5ebb6c25-Abstract-Datasets_and_Benchmarks.html"},{"key":"10915_CR69","doi-asserted-by":"publisher","unstructured":"Shan B, Yin W, Sun Y, Tian H, Wu H, Wang H (2022) Ernie-vil 2.0: Multi-view contrastive learning for image-text pre-training. CoRR https:\/\/doi.org\/10.48550\/arXiv.2209.15270arXiv:2209.15270","DOI":"10.48550\/arXiv.2209.15270"},{"key":"10915_CR70","unstructured":"Shen S, Li C, Hu X, Xie Y, Yang J, Zhang P, Gan Z, Wang L, Yuan L, Liu C, Keutzer K, Darrell T, Rohrbach A, Gao J (2022) K-LITE: learning transferable visual models with external knowledge. In: NeurIPS. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/63fef0802863f47775c3563e18cbba17-Abstract-Conference.html"},{"key":"10915_CR71","doi-asserted-by":"publisher","unstructured":"Shen S, Yang S, Zhang T, Zhai B, Gonzalez JE, Keutzer K, Darrell T (2022) Multitask vision-language prompt tuning. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.11720arXiv:2211.11720","DOI":"10.48550\/arXiv.2211.11720"},{"key":"10915_CR72","doi-asserted-by":"publisher","unstructured":"Shin T, Razeghi Y, IV RLL, Wallace E, Singh S (2020) Autoprompt: Eliciting knowledge from language models with automatically generated prompts. In: Webber, B., Cohn, T., He, Y., Liu, Y. (eds.) Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing, EMNLP 2020, Online, November 16-20, 2020, pp. 4222\u20134235. Association for Computational Linguistics, ??? . https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.346","DOI":"10.18653\/v1\/2020.emnlp-main.346"},{"key":"10915_CR73","unstructured":"Shu M, Nie W, Huang D, Yu Z, Goldstein T, Anandkumar A, Xiao C (2022) Test-time prompt tuning for zero-shot generalization in vision-language models. In: NeurIPS . http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/5bf2b802e24106064dc547ae9283bb0c-Abstract-Conference.html"},{"key":"10915_CR74","unstructured":"Sicilia A, Atwell K, Alikhani M, Hwang SJ (2022) Pac-bayesian domain adaptation bounds for multiclass learners. In: Cussens, J., Zhang, K. (eds.) Uncertainty in Artificial Intelligence, Proceedings of the Thirty-Eighth Conference on Uncertainty in Artificial Intelligence, UAI 2022, 1-5 August 2022, Eindhoven, The Netherlands. Proceedings of Machine Learning Research, vol. 180, pp. 1824\u20131834. PMLR, ??? . https:\/\/proceedings.mlr.press\/v180\/sicilia22a.html"},{"key":"10915_CR75","unstructured":"Soomro K, Zamir AR, Shah M (2012) UCF101: A dataset of 101 human actions classes from videos in the wild. CoRR arXiv:1212.0402"},{"key":"10915_CR76","doi-asserted-by":"publisher","unstructured":"Sun Q, Fang Y, Wu L, Wang X, Cao Y (2023) EVA-CLIP: improved training techniques for CLIP at scale. CoRR https:\/\/doi.org\/10.48550\/arXiv.2303.15389arXiv:2303.15389","DOI":"10.48550\/arXiv.2303.15389"},{"key":"10915_CR77","unstructured":"Tejankar A, Sanjabi M, Wu B, Xie S, Khabsa M, Pirsiavash H, Firooz H (2021) A fistful of words: Learning transferable visual models from bag-of-words supervision. CoRR arXiv:2112.13884"},{"key":"10915_CR78","doi-asserted-by":"publisher","unstructured":"Udandarao V, Gupta A, Albanie S (2022) Sus-x: Training-free name-only transfer of vision-language models. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.16198arXiv:2211.16198","DOI":"10.48550\/arXiv.2211.16198"},{"key":"10915_CR79","doi-asserted-by":"crossref","unstructured":"Vasu PKA, Pouransari H, Faghri F, Vemulapalli R, Tuzel O (2024) Mobileclip: Fast image-text models through multi-modal reinforced training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15963\u201315974","DOI":"10.1109\/CVPR52733.2024.01511"},{"key":"10915_CR80","doi-asserted-by":"publisher","unstructured":"Wang W, Bao H, Dong L, Bjorck J, Peng Z, Liu Q, Aggarwal K, Mohammed OK, Singhal S, Som S, Wei F (2022) Image as a foreign language: Beit pretraining for all vision and vision-language tasks. CoRR https:\/\/doi.org\/10.48550\/arXiv.2208.10442arXiv:2208.10442","DOI":"10.48550\/arXiv.2208.10442"},{"key":"10915_CR81","unstructured":"Wang H, Ge S, Lipton ZC, Xing EP (2019) Learning robust global representations by penalizing local predictive power. In: Wallach, H.M., Larochelle, H., Beygelzimer, A., d\u2019Alch\u00e9-Buc, F., Fox, E.B., Garnett, R. (eds.) Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada, pp. 10506\u201310518. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/3eefceb8087e964f89c2d59e8a249915-Abstract.html"},{"key":"10915_CR82","unstructured":"Wang J, Wang H, Deng J, Wu W, Zhang D (2021) Efficientclip: Efficient cross-modal pre-training by ensemble confident learning and language modeling. CoRR arXiv:2109.04699"},{"key":"10915_CR83","unstructured":"Wang Y, Yao Q (2019) Few-shot learning: A survey. CoRR arXiv:1904.05046"},{"key":"10915_CR84","unstructured":"Wang J, Zhang Y, Zhang L, Yang P, Gao X, Wu Z, Dong X, He J, Zhuo J, Yang Q, Huang Y, Li X, Wu Y, Lu J, Zhu X, Chen W, Han T, Pan K, Wang R, Wang H, Wu X, Zeng Z, Chen C, Gan R, Zhang J (2022) Fengshenbang 1.0: Being the foundation of chinese cognitive intelligence. CoRR abs\/2209.02970[SPACE]arXiv:2209.02970"},{"key":"10915_CR85","doi-asserted-by":"publisher","unstructured":"Wortsman M, Ilharco G, Kim JW, Li M, Kornblith S, Roelofs R, Lopes RG, Hajishirzi H, Farhadi A, Namkoong H, Schmidt L (2022) Robust fine-tuning of zero-shot models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 7949\u20137961. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52688.2022.00780","DOI":"10.1109\/CVPR52688.2022.00780"},{"key":"10915_CR86","doi-asserted-by":"publisher","unstructured":"Wu J, Li X, Wei C, Wang H, Yuille AL, Zhou Y, Xie C (2022) Unleashing the power of visual prompting at the pixel level. CoRR https:\/\/doi.org\/10.48550\/arXiv.2212.10556arXiv:2212.10556","DOI":"10.48550\/arXiv.2212.10556"},{"key":"10915_CR87","doi-asserted-by":"publisher","unstructured":"Xiao J, Hays J, Ehinger KA, Oliva A, Torralba A (2010) SUN database: Large-scale scene recognition from abbey to zoo. In: The Twenty-Third IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2010, San Francisco, CA, USA, 13-18 June 2010, pp. 3485\u20133492. IEEE Computer Society, ???. https:\/\/doi.org\/10.1109\/CVPR.2010.5539970","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"10915_CR88","doi-asserted-by":"crossref","unstructured":"Xie C-W, Sun S, Xiong X, Zheng Y, Zhao D, Zhou J (2023) Ra-clip: Retrieval augmented contrastive language-image pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19265\u201319274","DOI":"10.1109\/CVPR52729.2023.01846"},{"key":"10915_CR89","doi-asserted-by":"publisher","unstructured":"Xing Y, Shi Z, Meng Z, Lakemeyer G, Ma Y, Wattenhofer R (2021) KM-BART: knowledge enhanced multimodal BART for visual commonsense generation. In: Zong, C., Xia, F., Li, W., Navigli, R. (eds.) Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP 2021, (Volume 1: Long Papers), Virtual Event, August 1-6, 2021, pp. 525\u2013535. Association for Computational Linguistics, ???. https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.44","DOI":"10.18653\/v1\/2021.acl-long.44"},{"key":"10915_CR90","doi-asserted-by":"publisher","unstructured":"Xing Y, Wu Q, Cheng D, Zhang S, Liang G, Zhang Y (2022) Class-aware visual prompt tuning for vision-language pre-trained model. CoRR https:\/\/doi.org\/10.48550\/arXiv.2208.08340arXiv:2208.08340","DOI":"10.48550\/arXiv.2208.08340"},{"key":"10915_CR91","doi-asserted-by":"publisher","unstructured":"Yang Y, Panagopoulou A, Zhou S, Jin D, Callison-Burch C, Yatskar M (2022) Language in a bottle: Language model guided concept bottlenecks for interpretable image classification. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.11158arXiv:2211.11158","DOI":"10.48550\/arXiv.2211.11158"},{"key":"10915_CR92","doi-asserted-by":"publisher","unstructured":"Yang A, Pan J, Lin J, Men R, Zhang Y, Zhou J, Zhou C (2022) Chinese CLIP: contrastive vision-language pretraining in chinese. CoRR https:\/\/doi.org\/10.48550\/arXiv.2211.01335arXiv:2211.01335","DOI":"10.48550\/arXiv.2211.01335"},{"key":"10915_CR93","unstructured":"Yao L, Huang R, Hou L, Lu G, Niu M, Xu H, Liang X, Li Z, Jiang X, Xu C (2022) FILIP: fine-grained interactive language-image pre-training. In: The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022. OpenReview.net, ??? . https:\/\/openreview.net\/forum?id=cpDhcsEDC2"},{"key":"10915_CR94","doi-asserted-by":"publisher","unstructured":"Yao H, Zhang R, Xu C (2023) Visual-language prompt tuning with knowledge-guided context optimization. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2023, Vancouver, BC, Canada, June 17-24, 2023, pp. 6757\u20136767. IEEE, ???. https:\/\/doi.org\/10.1109\/CVPR52729.2023.00653","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"10915_CR95","doi-asserted-by":"crossref","unstructured":"Yu F, Tang J, Yin W, Sun Y, Tian H, Wu H, Wang H (2021) Ernie-vil: Knowledge enhanced vision-language representations through scene graphs. In: Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Thirty-Third Conference on Innovative Applications of Artificial Intelligence, IAAI 2021, The Eleventh Symposium on Educational Advances in Artificial Intelligence, EAAI 2021, Virtual Event, February 2-9, 2021, pp. 3208\u20133216. AAAI Press, ???. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/16431","DOI":"10.1609\/aaai.v35i4.16431"},{"key":"10915_CR96","doi-asserted-by":"publisher","unstructured":"Zang Y, Li W, Zhou K, Huang C, Loy CC (2022) Unified vision and language prompt learning. CoRR https:\/\/doi.org\/10.48550\/arXiv.2210.07225arXiv:2210.07225","DOI":"10.48550\/arXiv.2210.07225"},{"key":"10915_CR97","unstructured":"Zeng W, Ren X, Su T, Wang H, Liao Y, Wang Z, Jiang X, Yang Z, Wang, K, Zhang X, Li C, Gong Z, Yao, Y, Huang X, Wang J, Yu J, Guo Q, Yu Y, Zhang Y, Wang J, Tao H, Yan D, Yi Z, Peng F, Jiang F, Zhang H, Deng L, Zhang Y, Lin Z, Zhang C, Zhang S, Guo M, Gu S, Fan G, Wang Y, Jin X, Liu Q, Tian Y (2021) Pangu-$$\\alpha$$: Large-scale autoregressive pretrained chinese language models with auto-parallel computation. CoRR arXiv:abs\/2104.12369 )"},{"key":"10915_CR98","unstructured":"Zhang R, Fang R, Zhang W, Gao P, Li K, Dai J, Qiao Y, Li H (2021) Tip-adapter: Training-free clip-adapter for better vision-language modeling. CoRR arXiv:2111.03930"},{"key":"10915_CR99","doi-asserted-by":"publisher","unstructured":"Zhang Z, Han X, Liu Z, Jiang X, Sun M, Liu Q (2019) ERNIE: enhanced language representation with informative entities. In: Korhonen, A., Traum, D.R., M\u00e0rquez, L. (eds.) Proceedings of the 57th Conference of the Association for Computational Linguistics, ACL 2019, Florence, Italy, July 28- August 2, 2019, Volume 1: Long Papers, pp. 1441\u20131451. Association for Computational Linguistics, ??? . https:\/\/doi.org\/10.18653\/v1\/p19-1139","DOI":"10.18653\/v1\/p19-1139"},{"key":"10915_CR100","doi-asserted-by":"publisher","unstructured":"Zhang J, Huang J, Jin S, Lu S (2023) Vision-language models for vision tasks: A survey. CoRR https:\/\/doi.org\/10.48550\/arXiv.2304.00685arXiv:2304.00685","DOI":"10.48550\/arXiv.2304.00685"},{"key":"10915_CR101","doi-asserted-by":"publisher","unstructured":"Zhang B, Jin X, Gong W, Xu K, Zhang Z, Wang P, Shen X, Feng J (2023) Multimodal video adapter for parameter efficient video text retrieval. CoRR https:\/\/doi.org\/10.48550\/arXiv.2301.07868arXiv:2301.07868","DOI":"10.48550\/arXiv.2301.07868"},{"key":"10915_CR102","unstructured":"Zhang R, Qiu L, Zhang W, Zeng Z (2021) VT-CLIP: enhancing vision-language models with visual-guided texts. CoRR arXiv:2112.02399"},{"issue":"9","key":"10915_CR103","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Learning to prompt for vision-language models. Int J Comput Vis 130(9):2337\u20132348. https:\/\/doi.org\/10.1007\/s11263-022-01653-1","journal-title":"Int J Comput Vis"},{"key":"10915_CR104","doi-asserted-by":"publisher","unstructured":"Zhou C, Loy CC, Dai B (2022) Extract free dense labels from CLIP. In: Avidan, S., Brostow, G.J., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXVIII. Lecture Notes in Computer Science, vol. 13688, pp. 696\u2013712. Springer, ??? . https:\/\/doi.org\/10.1007\/978-3-031-19815-1_40","DOI":"10.1007\/978-3-031-19815-1_40"},{"key":"10915_CR105","doi-asserted-by":"publisher","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Conditional prompt learning for vision-language models. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022, pp. 16795\u201316804. IEEE, ??? . https:\/\/doi.org\/10.1109\/CVPR52688.2022.01631","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"10915_CR106","doi-asserted-by":"publisher","unstructured":"Zhu B, Niu Y, Han Y, Wu Y, Zhang H (2022) Prompt-aligned gradient for prompt tuning. CoRR https:\/\/doi.org\/10.48550\/arXiv.2205.14865arXiv:2205.14865","DOI":"10.48550\/arXiv.2205.14865"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-024-10915-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-024-10915-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-024-10915-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,24]],"date-time":"2024-09-24T23:42:13Z","timestamp":1727221333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-024-10915-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,27]]},"references-count":106,"journal-issue":{"issue":"10","published-online":{"date-parts":[[2024,10]]}},"alternative-id":["10915"],"URL":"https:\/\/doi.org\/10.1007\/s10462-024-10915-y","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-3497100\/v1","asserted-by":"object"}]},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,27]]},"assertion":[{"value":"12 August 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"268"}}