{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,11]],"date-time":"2025-10-11T00:24:20Z","timestamp":1760142260600,"version":"build-2065373602"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T00:00:00Z","timestamp":1751673600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T00:00:00Z","timestamp":1751673600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11263-025-02508-1","type":"journal-article","created":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T11:30:24Z","timestamp":1751715024000},"page":"6930-6952","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CAT-TPT: Class-Agnostic Text-based Test-time Prompt Tuning for Vision-Language Models"],"prefix":"10.1007","volume":"133","author":[{"given":"Youjia","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiling","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youngeun","family":"Kim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1774-9168","authenticated-orcid":false,"given":"Sungeun","family":"Hong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,5]]},"reference":[{"key":"2508_CR1","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., & Van\u00a0Gool, L. (2014). Food-101\u2013mining discriminative components with random forests. In Proc. of European Conf. on Computer Vision (ECCV), pp. 446\u2013461. Springer.","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"2508_CR2","unstructured":"Brown, T.B. (2020). Language models are few-shot learners. arXiv preprint arXiv:2005.14165."},{"issue":"4","key":"2508_CR3","doi-asserted-by":"publisher","first-page":"1108","DOI":"10.1007\/s11263-023-01904-9","volume":"132","author":"A Bulat","year":"2024","unstructured":"Bulat, A., & Tzimiropoulos, G. (2024). Language-aware soft prompting: Text-to-text optimization for few-and zero-shot adaptation of v & l models. International Journal of Computer Vision, 132(4), 1108\u20131125.","journal-title":"International Journal of Computer Vision"},{"key":"2508_CR4","unstructured":"Chen, Z., Duan, Y., Wang, W., He, J., Lu, T., Dai, J., & Qiao, Y. (2022). Vision transformer adapter for dense predictions. In Int. Conf. Learn. Represent."},{"key":"2508_CR5","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020). A simple framework for contrastive learning of visual representations, pp. 1597\u20131607. PMLR."},{"key":"2508_CR6","doi-asserted-by":"crossref","unstructured":"Cheung, T.-H., & Yeung, D.-Y. (2023). A survey of automated data augmentation for image classification: Learning to compose, mix, and generate. IEEE Trans. Neural Netw. Learn. Syst.","DOI":"10.1109\/TNNLS.2023.3282258"},{"key":"2508_CR7","doi-asserted-by":"crossref","unstructured":"Cho, E., Kim, J., & Kim, H.J. (2023). Distribution-aware prompt tuning for vision-language models. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 22004\u201322013.","DOI":"10.1109\/ICCV51070.2023.02011"},{"key":"2508_CR8","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., & Vedaldi, A. (2014). Describing textures in the wild. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 3606\u20133613.","DOI":"10.1109\/CVPR.2014.461"},{"key":"2508_CR9","doi-asserted-by":"crossref","unstructured":"Cubuk, E.D., Zoph, B., Mane, D., Vasudevan, V., & Le, Q.V. (2019). Autoaugment: Learning augmentation policies from data. IEEE Conf. Comput. Vis. Pattern Recog. (CVPR).","DOI":"10.1109\/CVPR.2019.00020"},{"key":"2508_CR10","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 248\u2013255. Ieee.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2508_CR11","first-page":"19822","volume":"34","author":"M Ding","year":"2021","unstructured":"Ding, M., Yang, Z., Hong, W., Zheng, W., Zhou, C., Yin, D., Lin, J., Zou, X., Shao, Z., Yang, H., et al. (2021). Cogview: Mastering text-to-image generation via transformers. Proc. of Neural Information Processing Systems (NeurIPS), 34, 19822\u201319835.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR12","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. In Int. Conf. Learn. Represent."},{"key":"2508_CR13","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2004). Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In IEEE Conf. Comput. Vis. Pattern Recog. Worksh. (CVPRW), pp. 178\u2013178. IEEE.","DOI":"10.1109\/CVPR.2004.383"},{"key":"2508_CR14","doi-asserted-by":"crossref","unstructured":"Feng, C.-M., Yu, K., Liu, Y., Khan, S., & Zuo, W. (2023). Diverse data augmentation with diffusions for effective test-time prompt tuning. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 2704\u20132714.","DOI":"10.1109\/ICCV51070.2023.00255"},{"key":"2508_CR15","doi-asserted-by":"crossref","unstructured":"Fuchs, C., Zanella, M., & De\u00a0Vleeschouwer, C. (2025). Online gaussian test-time adaptation of vision-language models. arXiv preprint arXiv:2501.04352.","DOI":"10.1109\/CVPRW67362.2025.00018"},{"key":"2508_CR16","unstructured":"Gal, R., Alaluf, Y., Atzmon, Y., Patashnik, O., Bermano, A.H., Chechik, G., & Cohen-or, D. (2022). An image is worth one word: Personalizing text-to-image generation using textual inversion. In Int. Conf. Learn. Represent."},{"issue":"2","key":"2508_CR17","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., & Qiao, Y. (2024). Clip-adapter: Better vision-language models with feature adapters. International Journal of Computer Vision, 132(2), 581\u2013595.","journal-title":"International Journal of Computer Vision"},{"key":"2508_CR18","doi-asserted-by":"crossref","unstructured":"Goyal, S., Kumar, A., Garg, S., Kolter, Z., & Raghunathan, A. (2023). Finetune like you pretrain: Improved finetuning of zero-shot vision models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 19338\u201319347.","DOI":"10.1109\/CVPR52729.2023.01853"},{"issue":"7","key":"2508_CR19","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., & Borth, D. (2019). Eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing, 12(7), 2217\u20132226.","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"key":"2508_CR20","unstructured":"Hendrycks, D., & Dietterich, T. (2019). Benchmarking neural network robustness to common corruptions and perturbations. Int. Conf. Learn. Represent."},{"key":"2508_CR21","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., Dorundo, E., Desai, R., Zhu, T., Parajuli, S., Guo, M., et al. (2021). The many faces of robustness: A critical analysis of out-of-distribution generalization. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 8340\u20138349.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"2508_CR22","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., & Song, D. (2021). Natural adversarial examples. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 15262\u201315271.","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"2508_CR23","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Proc. of Neural Information Processing Systems (NeurIPS), 33, 6840\u20136851.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR24","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685."},{"key":"2508_CR25","unstructured":"Huang, T., Chu, J., & Wei, F. (2022). Unsupervised prompt learning for vision-language models. arXiv preprint arXiv:2204.03649."},{"key":"2508_CR26","doi-asserted-by":"crossref","unstructured":"Huang, N., Zhang, Y., Tang, F., Ma, C., Huang, H., Dong, W., & Xu, C. (2024). Diffstyler: Controllable dual diffusion for text-driven image stylization. IEEE Trans. Neural Netw. Learn. Syst.","DOI":"10.1109\/TNNLS.2023.3342645"},{"key":"2508_CR27","unstructured":"Imam, R., Hanif, A., Zhang, J., Dawoud, K.W., Kementchedjhieva, Y., & Yaqub, M. (2025). Noise is an efficient learner for zero-shot vision-language models. arXiv preprint arXiv:2502.06019."},{"key":"2508_CR28","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.-C., Cardie, C., Belongie, S., Hariharan, B., & Lim, S.-N. (2022). Visual prompt tuning. In European Conference on Computer Vision, pp. 709\u2013727. Springer.","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"2508_CR29","doi-asserted-by":"crossref","unstructured":"Kalantidis, Y., Tolias, G., et al. (2024). Label propagation for zero-shot classification with vision-language models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 23209\u201323218.","DOI":"10.1109\/CVPR52733.2024.02190"},{"key":"2508_CR30","doi-asserted-by":"crossref","unstructured":"Karmanov, A., Guan, D., Lu, S., El\u00a0Saddik, A., & Xing, E. (2024). Efficient test-time adaptation of vision-language models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 14162\u201314171.","DOI":"10.1109\/CVPR52733.2024.01343"},{"key":"2508_CR31","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., & Khan, F.S. (2023). Maple: Multi-modal prompt learning. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 19113\u201319122.","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"2508_CR32","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Wasim, S.T., Naseer, M., Khan, S., Yang, M.-H., & Khan, F.S. (2023). Self-regulating prompts: Foundational model adaptation without forgetting. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 15190\u201315200.","DOI":"10.1109\/ICCV51070.2023.01394"},{"key":"2508_CR33","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., & Fei-Fei, L. (2013). 3d object representations for fine-grained categorization. In Proc. of Int\u2019l Conf. on Computer Vision Worksh. (ICCVW), pp. 554\u2013561.","DOI":"10.1109\/ICCVW.2013.77"},{"key":"2508_CR34","doi-asserted-by":"crossref","unstructured":"Lai, Z., Li, Z., Oliveira, L.C., Chauhan, J., Dugger, B.N., & Chuah, C.-N. (2023). Clipath: Fine-tune clip with visual feature fusion for pathology image analysis towards minimizing data collection efforts. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 2374\u20132380.","DOI":"10.1109\/ICCVW60793.2023.00251"},{"key":"2508_CR35","doi-asserted-by":"crossref","unstructured":"Li, P., Li, D., Li, W., Gong, S., Fu, Y., & Hospedales, T.M. (2021). A simple feature augmentation for domain generalization. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 8886\u20138895.","DOI":"10.1109\/ICCV48922.2021.00876"},{"key":"2508_CR36","unstructured":"Li, Y., Su, Y., Goodge, A., Jia, K., & Xu, X. (2025). Efficient and context-aware label propagation for zero-\/few-shot training-free adaptation of vision-language model. In Int. Conf. Learn. Represent."},{"key":"2508_CR37","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, F., Lim, S.-N., Belongie, S., & Weinberger, K.Q. (2021). On feature normalization and data augmentation. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 12383\u201312392.","DOI":"10.1109\/CVPR46437.2021.01220"},{"key":"2508_CR38","doi-asserted-by":"crossref","unstructured":"Li, B., Xu, X., Wang, X., Hou, Y., Feng, Y., Wang, F., Zhang, X., Zhu, Q., & Che, W. (2024). Semantic-guided generative image augmentation method with diffusion models for image classification. In AAAI, pp. 3018\u20133027.","DOI":"10.1609\/aaai.v38i4.28084"},{"key":"2508_CR39","doi-asserted-by":"crossref","unstructured":"Liao, C., Tsiligkaridis, T., & Kulis, B. (2024). Descriptor and word soups: Overcoming the parameter efficiency accuracy tradeoff for out-of-distribution few-shot learning. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 27015\u201327025.","DOI":"10.1109\/CVPR52733.2024.02551"},{"key":"2508_CR40","doi-asserted-by":"crossref","unstructured":"Long, S., Zhao, Z., Yuan, J., Tan, Z., Liu, J., Feng, J., Wang, S., & Wang, J. (2024). Mutual prompt leaning for vision language models. International Journal of Computer Vision, 1\u201319.","DOI":"10.1007\/s11263-024-02243-z"},{"key":"2508_CR41","unstructured":"Ma, X., Zhang, J., Guo, S., & Xu, W. (2023). Swapprompt: Test-time prompt adaptation for vision-language models. In Proc. of Neural Information Processing Systems (NeurIPS)."},{"key":"2508_CR42","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., & Vedaldi, A. (2013). Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151."},{"issue":"5","key":"2508_CR43","doi-asserted-by":"publisher","first-page":"1685","DOI":"10.1007\/s11263-023-01951-2","volume":"132","author":"X Mao","year":"2024","unstructured":"Mao, X., Chen, Y., Jia, X., Zhang, R., Xue, H., & Li, Z. (2024). Context-aware robust fine-tuning. International Journal of Computer Vision, 132(5), 1685\u20131700.","journal-title":"International Journal of Computer Vision"},{"key":"2508_CR44","unstructured":"Menon, S., & Vondrick, C. (2023). Visual classification via description from large language models. Int. Conf. Learn. Represent."},{"issue":"2","key":"2508_CR45","doi-asserted-by":"publisher","first-page":"596","DOI":"10.1007\/s11263-023-01895-7","volume":"132","author":"Y Ming","year":"2024","unstructured":"Ming, Y., & Li, Y. (2024). How does fine-tuning impact out-of-distribution detection for vision-language models? International Journal of Computer Vision, 132(2), 596\u2013609.","journal-title":"International Journal of Computer Vision"},{"key":"2508_CR46","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., & Chen, M. (2021). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741."},{"key":"2508_CR47","doi-asserted-by":"crossref","unstructured":"Nilsback, M.-E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. In 2008 Sixth Indian Conference on Computer Vision, Graphics & Image Processing, pp. 722\u2013729. IEEE.","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"2508_CR48","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., & Jawahar, C. (2012). Cats and dogs. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 3498\u20133505. IEEE.","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"2508_CR49","doi-asserted-by":"crossref","unstructured":"Pratt, S., Covert, I., Liu, R., & Farhadi, A. (2023). What does a platypus look like? generating customized prompts for zero-shot image classification. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 15691\u201315701.","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"2508_CR50","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al. (2021). Learning transferable visual models from natural language supervision, pp. 8748\u20138763. PMLR."},{"key":"2508_CR51","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., & Sutskever, I. (2021). Zero-shot text-to-image generation, pp. 8821\u20138831. PMLR."},{"key":"2508_CR52","unstructured":"Recht, B., Roelofs, R., Schmidt, L., & Shankar, V. (2019). Do imagenet classifiers generalize to imagenet?, pp. 5389\u20135400. PMLR."},{"key":"2508_CR53","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 10684\u201310695.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2508_CR54","doi-asserted-by":"crossref","unstructured":"Roth, K., Kim, J.M., Koepke, A., Vinyals, O., Schmid, C., & Akata, Z. (2023). Waffling around for performance: Visual classification with random words and broad concepts. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 15746\u201315757.","DOI":"10.1109\/ICCV51070.2023.01443"},{"key":"2508_CR55","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., & Aberman, K. (2023). Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 22500\u201322510.","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"2508_CR56","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Proc. of Neural Information Processing Systems (NeurIPS), 35, 36479\u201336494.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR57","unstructured":"Samadh, J.H.A., Gani, H., Hussein, N.H., Khattak, M.U., Naseer, M., Khan, F., & Khan, S. (2023). Align your prompts: Test-time prompting with distribution alignment for zero-shot generalization. In Proc. of Neural Information Processing Systems (NeurIPS)."},{"key":"2508_CR58","doi-asserted-by":"crossref","unstructured":"Shtedritski, A., Rupprecht, C., & Vedaldi, A. (2023). What does clip know about a red circle? visual prompt engineering for vlms. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 11987\u201311997.","DOI":"10.1109\/ICCV51070.2023.01101"},{"key":"2508_CR59","first-page":"14274","volume":"35","author":"M Shu","year":"2022","unstructured":"Shu, M., Nie, W., Huang, D.-A., Yu, Z., Goldstein, T., Anandkumar, A., & Xiao, C. (2022). Test-time prompt tuning for zero-shot generalization in vision-language models. Proc. of Neural Information Processing Systems (NeurIPS), 35, 14274\u201314289.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR60","first-page":"12533","volume":"34","author":"A Sinha","year":"2021","unstructured":"Sinha, A., Song, J., Meng, C., & Ermon, S. (2021). D2c: Diffusion-decoding models for few-shot conditional generation. Proc. of Neural Information Processing Systems (NeurIPS), 34, 12533\u201312548.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR61","first-page":"55361","volume":"36","author":"L Song","year":"2023","unstructured":"Song, L., Xue, R., Wang, H., Sun, H., Ge, Y., Shan, Y., et al. (2023). Meta-adapter: An online few-shot learner for vision-language model. Proc. of Neural Information Processing Systems (NeurIPS), 36, 55361\u201355374.","journal-title":"Proc. of Neural Information Processing Systems (NeurIPS)"},{"key":"2508_CR62","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402."},{"key":"2508_CR63","doi-asserted-by":"crossref","unstructured":"Sui, E., Wang, X., & Yeung-Levy, S. (2024). Just shift it: Test-time prototype shifting for zero-shot generalization with vision-language models. arXiv preprint arXiv:2403.12952.","DOI":"10.1109\/WACV61041.2025.00090"},{"key":"2508_CR64","unstructured":"Trabucco, B., Doherty, K., Gurinas, M.A., & Salakhutdinov, R. (2024). Effective data augmentation with diffusion models. Int. Conf. Learn. Represent."},{"key":"2508_CR65","doi-asserted-by":"crossref","unstructured":"Udandarao, V., Gupta, A., & Albanie, S. (2023). Sus-x: Training-free name-only transfer of vision-language models. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 2725\u20132736.","DOI":"10.1109\/ICCV51070.2023.00257"},{"key":"2508_CR66","unstructured":"Wang, H., Ge, S., Lipton, Z., & Xing, E. P. (2019). Learning robust global representations by penalizing local predictive power. Proc. of Neural Information Processing Systems (NeurIPS),32."},{"key":"2508_CR67","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 3485\u20133492. IEEE.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"2508_CR68","unstructured":"Xiao, Z., Yan, S., Hong, J., Cai, J., Jiang, X., Hu, Y., Shen, J., Wang, C., & Snoek, C.G. (2025). Dynaprompt: Dynamic test-time prompt tuning. In Int. Conf. Learn. Represent."},{"key":"2508_CR69","unstructured":"Xu, Y., Xie, L., Gu, X., Chen, X., Chang, H., Zhang, H., Chen, Z., Zhang, X., & Tian, Q. (2024). Qa-lora: Quantization-aware low-rank adaptation of large language models. In Int. Conf. Learn. Represent."},{"issue":"2","key":"2508_CR70","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1007\/s11263-024-02172-x","volume":"133","author":"C Xu","year":"2025","unstructured":"Xu, C., Zhu, Y., Shen, H., Chen, B., Liao, Y., Chen, X., & Wang, L. (2025). Progressive visual prompt learning with contrastive feature re-formation. International Journal of Computer Vision, 133(2), 511\u2013526.","journal-title":"International Journal of Computer Vision"},{"key":"2508_CR71","doi-asserted-by":"crossref","unstructured":"Yao, H., Zhang, R., & Xu, C. (2024). Tcp: Textual-based class-aware prompt tuning for visual-language model. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 23438\u201323448.","DOI":"10.1109\/CVPR52733.2024.02212"},{"key":"2508_CR72","unstructured":"Yoon, H.S., Yoon, E., Tee, J.T.J., Hasegawa-Johnson, M.A., Li, Y., & Yoo, C.D. (2024). C-TPT: Calibrated test-time prompt tuning for vision-language models via text feature dispersion. In Int. Conf. Learn. Represent."},{"key":"2508_CR73","doi-asserted-by":"crossref","unstructured":"Yun, S., Han, D., Oh, S.J., Chun, S., Choe, J., & Yoo, Y. (2019). Cutmix: Regularization strategy to train strong classifiers with localizable features. In Proc. of Int\u2019l Conf. on Computer Vision (ICCV), pp. 6023\u20136032.","DOI":"10.1109\/ICCV.2019.00612"},{"key":"2508_CR74","doi-asserted-by":"crossref","unstructured":"Zanella, M., & Ben\u00a0Ayed, I. (2024). On the test-time zero-shot generalization of vision-language models: Do we really need prompt learning? In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 23783\u201323793.","DOI":"10.1109\/CVPR52733.2024.02245"},{"key":"2508_CR75","unstructured":"Zhang, Q., Chen, M., Bukharin, A., He, P., Cheng, Y., Chen, W., & Zhao, T. (2024). Adaptive budget allocation for parameter-efficient fine-tuning. In Int. Conf. Learn. Represent."},{"key":"2508_CR76","unstructured":"Zhang, H., Cisse, M., Dauphin, Y.N., & Lopez-Paz, D. (2018). mixup: Beyond empirical risk minimization. In Int. Conf. Learn. Represent."},{"key":"2508_CR77","unstructured":"Zhang, C., Stepputtis, S., Sycara, K., & Xie, Y. (2024). Dual prototype evolving for test-time generalization of vision-language models. Proc. of Neural Information Processing Systems (NeurIPS)."},{"key":"2508_CR78","unstructured":"Zhang, T., Wang, J., Guo, H., Dai, T., Chen, B., & Xia, S.-T. (2024). Boostadapter: Improving vision-language test-time adaptation via regional bootstrapping. Proc. of Neural Information Processing Systems (NeurIPS)."},{"key":"2508_CR79","doi-asserted-by":"crossref","unstructured":"Zhang, R., Zhang, W., Fang, R., Gao, P., Li, K., Dai, J., Qiao, Y., & Li, H. (2022). Tip-adapter: Training-free adaption of clip for few-shot classification. In Proc. of European Conf. on Computer Vision (ECCV), pp. 493\u2013510. Springer.","DOI":"10.1007\/978-3-031-19833-5_29"},{"key":"2508_CR80","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zhu, W., Tang, H., Ma, Z., Zhou, K., & Zhang, L. (2024). Dual memory networks: A versatile adaptation approach for vision-language models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR), pp. 28718\u201328728.","DOI":"10.1109\/CVPR52733.2024.02713"},{"key":"2508_CR81","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022). Conditional prompt learning for vision-language models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR)","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"2508_CR82","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022). Learning to prompt for vision-language models. International Journal of Computer Vision.","DOI":"10.1007\/s11263-022-01653-1"},{"key":"2508_CR83","doi-asserted-by":"crossref","unstructured":"Zhou, L., Ye, M., Li, S., Li, N., Zhu, X., Deng, L., Liu, H., & Lei, Z. (2025). Bayesian test-time adaptation for vision-language models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR).","DOI":"10.1109\/CVPR52734.2025.02792"},{"key":"2508_CR84","unstructured":"Zhu, Y., Ji, Y., Zhao, Z., Wu, G., & Wang, L. (2024). Awt: Transferring vision-language models via augmentation, weighting, and transportation. In Proc. of Neural Information Processing Systems (NeurIPS)."},{"key":"2508_CR85","unstructured":"Zhu, Y., Zhang, G., Xu, C., Shen, H., Chen, X., Wu, G., & Wang, L. (2024). Efficient test-time prompt tuning for vision-language models. arXiv preprint arXiv:2408.05775."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02508-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02508-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02508-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,10]],"date-time":"2025-10-10T08:53:26Z","timestamp":1760086406000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02508-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,5]]},"references-count":85,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["2508"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02508-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"type":"print","value":"0920-5691"},{"type":"electronic","value":"1573-1405"}],"subject":[],"published":{"date-parts":[[2025,7,5]]},"assertion":[{"value":"6 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 June 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 July 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}