{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T18:28:16Z","timestamp":1784399296494,"version":"3.55.0"},"reference-count":70,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,9,15]],"date-time":"2023-09-15T00:00:00Z","timestamp":1694736000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,15]],"date-time":"2023-09-15T00:00:00Z","timestamp":1694736000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,2]]},"DOI":"10.1007\/s11263-023-01891-x","type":"journal-article","created":{"date-parts":[[2023,9,15]],"date-time":"2023-09-15T05:02:11Z","timestamp":1694754131000},"page":"581-595","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":963,"title":["CLIP-Adapter: Better Vision-Language Models with Feature Adapters"],"prefix":"10.1007","volume":"132","author":[{"given":"Peng","family":"Gao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shijie","family":"Geng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Renrui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Teli","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongyao","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongfeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,9,15]]},"reference":[{"key":"1891_CR1","volume-title":"Advances in neural information processing systems","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J. B., Donahue, J., Luc, P., et al. (2022). Flamingo: a visual language model for few-shot learning. In A. H. Oh, A. Agarwal, D. Belgrave, et al. (Eds.), Advances in neural information processing systems. MIT Press."},{"key":"1891_CR2","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., & Buehler, C., et al. (2018). Bottom-up and top-down attention for image captioning and visual question answering. In CVPR.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"1891_CR3","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., & Van Gool, L. (2014). Food-101\u2013mining discriminative components with random forests. In European conference on computer vision, Springer, pp. 446\u2013461.","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"1891_CR4","unstructured":"Brown, T., Mann, B., & Ryder, N., et al. (2020). Language models are few-shot learners. In NeurIPS."},{"key":"1891_CR5","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., & Synnaeve, G., et al. (2020). End-to-end object detection with transformers. In ECCV.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1891_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Y. C., Li, L., & Yu, L., et al. (2020). Uniter: Learning universal image-text representations. In ECCV.","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"1891_CR7","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., & Kokkinos, I., et al. (2014). Describing textures in the wild. In Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3606\u20133613.","DOI":"10.1109\/CVPR.2014.461"},{"key":"1891_CR8","doi-asserted-by":"crossref","unstructured":"Conneau, A., Khandelwal, K., & Goyal, N., et al. (2020). Unsupervised cross-lingual representation learning at scale. In ACL.","DOI":"10.18653\/v1\/2020.acl-main.747"},{"key":"1891_CR9","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., & Socher, R., et al. (2009). Imagenet: A large-scale hierarchical image database. In CVPR.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1891_CR10","unstructured":"Devlin, J., Chang, M. W., & Lee, K., et al. (2019). Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL-HLT."},{"key":"1891_CR11","unstructured":"Dong, L., Yang, N., & Wang, W., et al. (2019). Unified language model pre-training for natural language understanding and generation. In NeurIPS."},{"key":"1891_CR12","unstructured":"Dosovitskiy, A., Beyer, L., & Kolesnikov, A., et al. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In ICLR."},{"key":"1891_CR13","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2004). Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In 2004 conference on computer vision and pattern recognition workshop, IEEE, p. 178.","DOI":"10.1109\/CVPR.2004.383"},{"key":"1891_CR14","doi-asserted-by":"crossref","unstructured":"Gao, P., Jiang, Z., & You, H., et al. (2019). Dynamic fusion with intra-and inter-modality attention flow for visual question answering. In CVPR.","DOI":"10.1109\/CVPR.2019.00680"},{"key":"1891_CR15","unstructured":"Gao, P., Lu, J., & Li, H., et al. (2021a). Container: Context aggregation network. In NeurIPS."},{"key":"1891_CR16","doi-asserted-by":"crossref","unstructured":"Gao, P., Zheng, M., & Wang, X., et al. (2021b) Fast convergence of detr with spatially modulated co-attention. In Proceedings of the IEEE\/CVF international conference on computer vision, pp 3621\u20133630.","DOI":"10.1109\/ICCV48922.2021.00360"},{"key":"1891_CR17","doi-asserted-by":"crossref","unstructured":"Gao, T., Fisch, A., & Chen, D. (2021c). Making pre-trained language models better few-shot learners. In ACL-IJCNLP.","DOI":"10.18653\/v1\/2021.acl-long.295"},{"key":"1891_CR18","doi-asserted-by":"crossref","unstructured":"Gu, Y., Han, X., & Liu, Z., et al. (2022). Ppt: Pre-trained prompt tuning for few-shot learning. In Proceedings of the 60th annual meeting of the association for computational linguistics (Volume 1: Long Papers), pp. 8410\u20138423.","DOI":"10.18653\/v1\/2022.acl-long.576"},{"key":"1891_CR19","unstructured":"He, J., Zhou, C., & Ma, X., et al. (2022). Towards a unified view of parameter-efficient transfer learning. In International conference on learning representations."},{"key":"1891_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., & Ren, S., et al. (2016). Deep residual learning for image recognition. In CVPR.","DOI":"10.1109\/CVPR.2016.90"},{"issue":"7","key":"1891_CR21","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., et al. (2019). Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing, 12(7), 2217\u20132226.","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"1891_CR22","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., et al. (2021a). The many faces of robustness: A critical analysis of out-of-distribution generalization. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8340\u20138349.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"1891_CR23","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., et al. (2021b). Natural adversarial examples. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 15,262\u201315,271.","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"1891_CR24","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., et al. (2019). Parameter-efficient transfer learning for nlp. In International conference on machine learning, PMLR, pp. 2790\u20132799."},{"key":"1891_CR25","unstructured":"Howard, A. G., Zhu, M., Chen, B., et al. (2017). Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861"},{"key":"1891_CR26","unstructured":"Hu, S., Zhang, Z., Ding, N., et al. (2022). Sparse structure search for parameter-efficient tuning. arXiv preprint arXiv:2206.07382"},{"key":"1891_CR27","unstructured":"Jia, C., Yang, Y., Xia, Y., et al. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. In ICML."},{"key":"1891_CR28","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B. C., et al. (2022). Visual prompt tuning. In ECCV, pp. 709\u2013727.","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"1891_CR29","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1162\/tacl_a_00324","volume":"8","author":"Z Jiang","year":"2020","unstructured":"Jiang, Z., Xu, F. F., Araki, J., et al. (2020). How can we know what language models know? Transactions of the Association for Computational Linguistics, 8, 423\u2013438.","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"1891_CR30","unstructured":"Kim, J. H., Jun, J., & Zhang, B. T. (2018). Bilinear attention networks. In NIPS."},{"key":"1891_CR31","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., et al. (2013). 3d object representations for fine-grained categorization. In Proceedings of the IEEE international conference on computer vision workshops, pp. 554\u2013561.","DOI":"10.1109\/ICCVW.2013.77"},{"key":"1891_CR32","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In NIPS."},{"key":"1891_CR33","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., & Constant, N. (2021). The power of scale for parameter-efficient prompt tuning. In EMNLP.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"1891_CR34","unstructured":"Li, C., Liu, H., Li, L. H., et al. (2022). ELEVATER: A benchmark and toolkit for evaluating language-augmented visual models. In Thirty-sixth conference on neural information processing systems datasets and benchmarks track."},{"key":"1891_CR35","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., et al. (2021). Align before fuse: Vision and language representation learning with momentum distillation. Advances in Neural Information Processing Systems, 34, 9694\u20139705.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"1891_CR36","doi-asserted-by":"crossref","unstructured":"Li, X., Yin, X., Li, C., et al. (2020). Oscar: Object-semantics aligned pre-training for vision-language tasks. In ECCV.","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"1891_CR37","doi-asserted-by":"crossref","unstructured":"Li, X. L., & Liang, P. (2021). Prefix-tuning: Optimizing continuous prompts for generation. In ACL.","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"1891_CR38","first-page":"109","volume":"35","author":"D Lian","year":"2022","unstructured":"Lian, D., Zhou, D., Feng, J., et al. (2022). Scaling and shifting your features: A new baseline for efficient model tuning. Advances in Neural Information Processing Systems, 35, 109\u2013123.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"9","key":"1891_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3560815","volume":"55","author":"P Liu","year":"2023","unstructured":"Liu, P., Yuan, W., Fu, J., et al. (2023). Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. ACM Computing Surveys, 55(9), 1\u201335.","journal-title":"ACM Computing Surveys"},{"key":"1891_CR40","unstructured":"Liu, X., Zheng, Y., Du, Z., et al. (2021). Gpt understands, too. arXiv preprint arXiv:2103.10385"},{"key":"1891_CR41","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., & Darrell, T. (2015). Fully convolutional networks for semantic segmentation. In CVPR.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"1891_CR42","unstructured":"Lu, J., Batra, D., Parikh, D., et al. (2019). Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In NeurIPS."},{"key":"1891_CR43","unstructured":"Maji, S., Rahtu, E., Kannala, J., et al. (2013). Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151."},{"key":"1891_CR44","first-page":"25,346","volume":"34","author":"M Mao","year":"2021","unstructured":"Mao, M., Zhang, R., Zheng, H., et al. (2021). Dual-stream network for visual recognition. Advances in Neural Information Processing Systems, 34, 25,346-25,358.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"1891_CR45","doi-asserted-by":"crossref","unstructured":"Nilsback, M. E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. IEEE: Graphics and 008 sixth Indian conference on computer vision (pp. 722\u2013729).","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"1891_CR46","doi-asserted-by":"crossref","unstructured":"Parkhi, O. M., Vedaldi, A., Zisserman, A., et al. (2012). Cats and dogs. In 2012 IEEE conference on computer vision and pattern recognition, IEEE, pp 3498\u20133505.","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"1891_CR47","unstructured":"Radford, A., Wu, J., Child, R., et al. (2019). Language models are unsupervised multitask learners. OpenAI blog."},{"key":"1891_CR48","unstructured":"Radford, A., Kim, J. W., Hallacy, C., et al. (2021). Learning transferable visual models from natural language supervision. In International conference on machine learning, PMLR, pp. 8748\u20138763."},{"key":"1891_CR49","unstructured":"Recht, B., Roelofs, R., Schmidt, L., et al. (2019). Do imagenet classifiers generalize to imagenet? In International conference on machine learning, PMLR, pp. 5389\u20135400."},{"key":"1891_CR50","unstructured":"Ren, S., He, K., Girshick, R., et al. (2015). Faster r-cnn: Towards real-time object detection with region proposal networks. In NIPS."},{"key":"1891_CR51","doi-asserted-by":"crossref","unstructured":"Shin, T., Razeghi, Y, Logan IV, R. L., et al. (2020). Autoprompt: Eliciting knowledge from language models with automatically generated prompts. In EMNLP.","DOI":"10.18653\/v1\/2020.emnlp-main.346"},{"key":"1891_CR52","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition. In ICLR."},{"key":"1891_CR53","unstructured":"Soomro, K., Zamir, A. R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402."},{"key":"1891_CR54","unstructured":"Sun, T., Shao, Y., Qian, H., et al. (2022). Black-box tuning for language-model-as-a-service. In International conference on machine learning, PMLR, pp. 20,841\u201320,855."},{"key":"1891_CR55","first-page":"12,991","volume":"35","author":"YL Sung","year":"2022","unstructured":"Sung, Y. L., Cho, J., & Bansal, M. (2022). Lst: Ladder side-tuning for parameter and memory efficient transfer learning. Advances in Neural Information Processing Systems, 35, 12,991-13,005.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"1891_CR56","doi-asserted-by":"crossref","unstructured":"Sung, Y. L., Cho, J., & Bansal, M. (2022b). Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 5227\u20135237.","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"1891_CR57","doi-asserted-by":"crossref","unstructured":"Tan, H., & Bansal, M. (2019). Lxmert: Learning cross-modality encoder representations from transformers. In EMNLP-IJCNLP.","DOI":"10.18653\/v1\/D19-1514"},{"key":"1891_CR58","unstructured":"Touvron, H., Cord, M., Douze, M., et al. (2021). Training data-efficient image transformers and distillation through attention. In ICML."},{"key":"1891_CR59","first-page":"200","volume":"34","author":"M Tsimpoukelli","year":"2021","unstructured":"Tsimpoukelli, M., Menick, J. L., Cabi, S., et al. (2021). Multimodal few-shot learning with frozen language models. Advances in Neural Information Processing Systems, 34, 200\u2013212.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"1891_CR60","unstructured":"Van der Maaten, L., & Hinton, G. (2008). Visualizing data using t-sne. Journal of machine learning research 9(11)."},{"key":"1891_CR61","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al. (2017). Attention is all you need. In NIPS."},{"key":"1891_CR62","unstructured":"Wang, H., Ge, S., Lipton, Z., et al. (2019). Learning robust global representations by penalizing local predictive power. Advances in Neural Information Processing Systems, 32."},{"key":"1891_CR63","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., Dong, L., et al. (2022a). Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442.","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"1891_CR64","unstructured":"Wang, Z., Yu, J., Yu, A. W., et al. (2022b). SimVLM: Simple visual language model pretraining with weak supervision. In International Conference on Learning Representations."},{"key":"1891_CR65","doi-asserted-by":"crossref","unstructured":"Wortsman, M., Ilharco, G., Kim, J. W., et al. (2022). Robust fine-tuning of zero-shot models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7959\u20137971.","DOI":"10.1109\/CVPR52688.2022.00780"},{"key":"1891_CR66","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K. A., et al. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In 2010 IEEE computer society conference on computer vision and pattern recognition, IEEE, pp. 3485\u20133492.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"1891_CR67","doi-asserted-by":"crossref","unstructured":"Yao, Y., Zhang, A., Zhang, Z., et al. (2021). Cpt: Colorful prompt tuning for pre-trained vision-language models. arXiv preprint arXiv:2109.11797","DOI":"10.18653\/v1\/2022.findings-acl.273"},{"key":"1891_CR68","doi-asserted-by":"crossref","unstructured":"Yao, Y., Chen, Q., Zhang, A., et al. (2022). PEVL: Position-enhanced pre-training and prompt tuning for vision-language models. In Proceedings of the 2022 conference on empirical methods in natural language processing, pp. 11,104\u201311,117.","DOI":"10.18653\/v1\/2022.emnlp-main.763"},{"key":"1891_CR69","doi-asserted-by":"crossref","unstructured":"Yu, Z., Yu, J., Cui, Y., et al. (2019). Deep modular co-attention networks for visual question answering. In CVPR.","DOI":"10.1109\/CVPR.2019.00644"},{"key":"1891_CR70","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C. C., et al. (2022). Learning to prompt for vision-language models. International Journal of Computer Vision, pp. 1\u201312.","DOI":"10.1007\/s11263-022-01653-1"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01891-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-023-01891-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01891-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,30]],"date-time":"2024-01-30T07:21:45Z","timestamp":1706599305000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-023-01891-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,15]]},"references-count":70,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,2]]}},"alternative-id":["1891"],"URL":"https:\/\/doi.org\/10.1007\/s11263-023-01891-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,15]]},"assertion":[{"value":"4 October 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 August 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 September 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 October 2023","order":4,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Update","order":5,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The original version has been revised to update Fig. 1","order":6,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}}]}}