{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T17:03:07Z","timestamp":1784566987443,"version":"3.55.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T00:00:00Z","timestamp":1777680000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T00:00:00Z","timestamp":1777680000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406037"],"award-info":[{"award-number":["62406037"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62076032"],"award-info":[{"award-number":["62076032"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372150"],"award-info":[{"award-number":["62372150"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11263-026-02851-x","type":"journal-article","created":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T17:03:27Z","timestamp":1777741407000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Plug-and-play Class-aware Knowledge Injection for Prompt Learning with Visual-Language Model"],"prefix":"10.1007","volume":"134","author":[{"given":"Junhui","family":"Yin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2179-8301","authenticated-orcid":false,"given":"Nan","family":"Pu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingfeng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaojie","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhun","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,2]]},"reference":[{"key":"2851_CR1","doi-asserted-by":"crossref","unstructured":"Abdul\u00a0Samadh, J., Gani, M.H., Hussein, N., Khattak, M.U., Naseer, M.M., Shahbaz\u00a0Khan, F., & Khan, S.H. (2024). Align your prompts: Test-time prompting with distribution alignment for zero-shot generalization. Adv. Neural Inform. Process. Syst. 36.","DOI":"10.52202\/075280-3525"},{"key":"2851_CR2","unstructured":"Bahng, H., Jahanian, A., Sankaranarayanan, S., & Isola, P. (2022). Exploring visual prompts for adapting large-scale models. arXiv preprint arXiv:2203.17274."},{"key":"2851_CR3","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., & Van\u00a0Gool, L. (2014). Food-101\u2013mining discriminative components with random forests. In: Eur. Conf. Comput. Vis., pp. 446\u2013461. Springer","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"2851_CR4","unstructured":"Chen, G., Yao, W., Song, X., Li, X., Rao, Y., & Zhang, K. (2022). Plot: Prompt learning with optimal transport for vision-language models. arXiv preprint arXiv:2210.01253."},{"key":"2851_CR5","unstructured":"Chi, Z., Gu, L., Liu, H., Wang, Z., Wu, Y., Wang, Y., & Plataniotis, K.N. (2025). Learning to adapt frozen clip for few-shot test-time domain adaptation. arXiv preprint arXiv:2506.17307"},{"key":"2851_CR6","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., & Vedaldi, A. (2014). Describing textures in the wild. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 3606\u20133613.","DOI":"10.1109\/CVPR.2014.461"},{"key":"2851_CR7","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2004). Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 178\u2013178. IEEE","DOI":"10.1109\/CVPR.2004.383"},{"key":"2851_CR8","doi-asserted-by":"crossref","unstructured":"Feng, C.-M., Yu, K., Liu, Y., Khan, S., & Zuo, W. (2023). Diverse data augmentation with diffusions for effective test-time prompt tuning. In: Int. Conf. Comput. Vis.","DOI":"10.1109\/ICCV51070.2023.00255"},{"issue":"9","key":"2851_CR9","doi-asserted-by":"publisher","first-page":"3770","DOI":"10.1007\/s11263-024-02026-6","volume":"132","author":"V Gabeff","year":"2024","unstructured":"Gabeff, V., Ru\u00dfwurm, M., Tuia, D., & Mathis, A. (2024). Wildclip: Scene and animal attribute retrieval from camera trap data with domain-adapted vision-language models. Int. J. Comput. Vis., 132(9), 3770\u20133786.","journal-title":"Int. J. Comput. Vis."},{"key":"2851_CR10","volume-title":"Clip-adapter: Better vision-language models with feature adapters","author":"P Gao","year":"2024","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., & Qiao, Y. (2024). Clip-adapter: Better vision-language models with feature adapters. Vis: Int. J. Comput."},{"key":"2851_CR11","unstructured":"Grave, E., Cisse, M.M., & Joulin, A. (2017). Unbounded cache model for online language modeling with open vocabulary. Adv. Neural Inform. Process. Syst. 30."},{"key":"2851_CR12","unstructured":"He, J., Zhou, C., Ma, X., Berg-Kirkpatrick, T., & Neubig, G. (2021). Towards a unified view of parameter-efficient transfer learning. arXiv preprint arXiv:2110.04366."},{"issue":"7","key":"2851_CR13","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., & Borth, D. (2019). Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing, 12(7), 2217\u20132226.","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"2851_CR14","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.-C., Cardie, C., Belongie, S., Hariharan, B., & Lim, S.-N. (2022). Visual prompt tuning. In: Eur. Conf. Comput. Vis., pp. 709\u2013727. Springer","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"2851_CR15","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.-T., Parekh, Z., Pham, H., Le, Q., Sung, Y.-H., Li, Z., & Duerig, T. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. In: Int. Conf. Mach. Learn., pp. 4904\u20134916. PMLR"},{"key":"2851_CR16","doi-asserted-by":"crossref","unstructured":"Karmanov, A., Guan, D., Lu, S., El\u00a0Saddik, A., & Xing, E. (2024). Efficient test-time adaptation of vision-language models. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 14162\u201314171.","DOI":"10.1109\/CVPR52733.2024.01343"},{"key":"2851_CR17","doi-asserted-by":"crossref","unstructured":"Khandelwal, A. (2024). Promptsync: Bridging domain gaps in vision-language models through class-aware prototype alignment and discrimination. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 7819\u20137828.","DOI":"10.1109\/CVPRW63382.2024.00778"},{"key":"2851_CR18","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., & Khan, F.S. (2023). Maple: Multi-modal prompt learning. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 19113\u201319122.","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"2851_CR19","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Wasim, S.T., Naseer, M., Khan, S., Yang, M.-H., & Khan, F.S. (2023). Self-regulating prompts: Foundational model adaptation without forgetting. In: Int. Conf. Comput. Vis., pp. 15190\u201315200.","DOI":"10.1109\/ICCV51070.2023.01394"},{"key":"2851_CR20","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., & Fei-Fei, L. (2013). 3d object representations for fine-grained categorization. In: IEEE Conf. Comput. Vis. Pattern Recog. Worksh., pp. 554\u2013561.","DOI":"10.1109\/ICCVW.2013.77"},{"key":"2851_CR21","doi-asserted-by":"crossref","unstructured":"Lafon, M., Ramzi, E., Rambour, C., Audebert, N., & Thome, N. (2024). Gallop: Learning global and local prompts for vision-language models. In: Eur. Conf. Comput. Vis., pp. 264\u2013282. Springer","DOI":"10.1007\/978-3-031-73030-6_15"},{"key":"2851_CR22","doi-asserted-by":"publisher","first-page":"1008","DOI":"10.52202\/068431-0074","volume":"35","author":"J Lee","year":"2022","unstructured":"Lee, J., Kim, J., Shon, H., Kim, B., Kim, S. H., Lee, H., & Kim, J. (2022). Uniclip: Unified framework for cocoopntrastive language-image pre-training. Adv. Neural Inform. Process. Syst., 35, 1008\u20131019.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2851_CR23","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., & Constant, N. (2021). The power of scale for parameter-efficient prompt tuning. In: EMNLP.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"2851_CR24","doi-asserted-by":"crossref","unstructured":"Li, X.L., Liang, P. (2021). Prefix-tuning: Optimizing continuous prompts for generation. In: ACL.","DOI":"10.18653\/v1\/2021.acl-long.353"},{"issue":"1","key":"2851_CR25","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1007\/s11263-024-02181-w","volume":"133","author":"J Liang","year":"2025","unstructured":"Liang, J., He, R., & Tan, T. (2025). A comprehensive survey on test-time adaptation under distribution shifts. Int. J. Comput. Vis., 133(1), 31\u201364.","journal-title":"Int. J. Comput. Vis."},{"issue":"9","key":"2851_CR26","first-page":"1","volume":"55","author":"P Liu","year":"2023","unstructured":"Liu, P., Yuan, W., Fu, J., Jiang, Z., Hayashi, H., & Neubig, G. (2023). Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. ACM Computing Surveys, 55(9), 1\u201335.","journal-title":"ACM Computing Surveys"},{"key":"2851_CR27","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., & Vedaldi, A. (2013). Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151."},{"key":"2851_CR28","unstructured":"Merity, S., Xiong, C., Bradbury, J., & Socher, R. (2016). Pointer sentinel mixture models. arXiv preprint arXiv:1609.07843."},{"key":"2851_CR29","doi-asserted-by":"publisher","first-page":"76298","DOI":"10.52202\/075280-3335","volume":"36","author":"A Miyai","year":"2023","unstructured":"Miyai, A., Yu, Q., Irie, G., & Aizawa, K. (2023). Locoop: Few-shot out-of-distribution detection via prompt learning. Adv. Neural Inform. Process. Syst., 36, 76298\u201376310.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2851_CR30","doi-asserted-by":"crossref","unstructured":"Nilsback, M.-E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. In: 2008 Sixth Indian Conference on Computer Vision, Graphics & Image Processing, pp. 722\u2013729. IEEE","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"2851_CR31","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., & Jawahar, C. (2012). Cats and dogs. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 3498\u20133505. IEEE","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"2851_CR32","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J., et al. (2021). Learning transferable visual models from natural language supervision. In: Int. Conf. Mach. Learn., pp. 8748\u20138763 . PMLR"},{"key":"2851_CR33","doi-asserted-by":"publisher","first-page":"14274","DOI":"10.52202\/068431-1038","volume":"35","author":"M Shu","year":"2022","unstructured":"Shu, M., Nie, W., Huang, D.-A., Yu, Z., Goldstein, T., Anandkumar, A., & Xiao, C. (2022). Test-time prompt tuning for zero-shot generalization in vision-language models. Adv. Neural Inform. Process. Syst., 35, 14274\u201314289.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2851_CR34","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402."},{"key":"2851_CR35","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhang, Z., Lee, C.-Y., Zhang, H., Sun, R., Ren, X., Su, G., Perot, V., Dy, J., & Pfister, T. (2022). Learning to prompt for continual learning. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 139\u2013149.","DOI":"10.1109\/CVPR52688.2022.00024"},{"key":"2851_CR36","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 3485\u20133492. IEEE","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"2851_CR37","doi-asserted-by":"crossref","unstructured":"Yao, H., Zhang, R., & Xu, C. (2024). Tcp: Textual-based class-aware prompt tuning for visual-language model. In: IEEE Conf. Comput. Vis. Pattern Recog., pp. 23438\u201323448.","DOI":"10.1109\/CVPR52733.2024.02212"},{"key":"2851_CR38","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., & Loy, C.C. (2022). Unified vision and language prompt learning. arXiv preprint arXiv:2210.07225."},{"key":"2851_CR39","unstructured":"Zhang, R., Fang, R., Zhang, W., Gao, P., Li, K., Dai, J., Qiao, Y., & Li, H. (2021). Tip-adapter: Training-free clip-adapter for better vision-language modeling. arXiv preprint arXiv:2111.03930."},{"key":"2851_CR40","volume-title":"Wang, Yang: Promptkd: Unsupervised prompt distillation for vision-language models","author":"L Zheng","year":"2024","unstructured":"Zheng, L., Xiang, L., Xinyi, F., Xing, Z., & Weiqiang, J. (2024). Wang, Yang: Promptkd: Unsupervised prompt distillation for vision-language models. Pattern Recog: IEEE Conf. Comput. Vis."},{"key":"2851_CR41","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022). Conditional prompt learning for vision-language models. In: IEEE Conf. Comput. Vis. Pattern Recog.","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"2851_CR42","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1","volume-title":"Learning to prompt for vision-language models","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. Vis: Int. J. Comput."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02851-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02851-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02851-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T16:14:17Z","timestamp":1784564057000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02851-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,2]]},"references-count":42,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2851"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02851-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,2]]},"assertion":[{"value":"27 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"254"}}