{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T08:09:19Z","timestamp":1781165359930,"version":"3.54.1"},"reference-count":72,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T00:00:00Z","timestamp":1768003200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T00:00:00Z","timestamp":1768003200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004837","name":"Ministerio de Ciencia e Innovaci\u00f3n","doi-asserted-by":"publisher","award":["PID2022-140189OB-C22"],"award-info":[{"award-number":["PID2022-140189OB-C22"]}],"id":[{"id":"10.13039\/501100004837","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s11263-025-02630-0","type":"journal-article","created":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T05:15:11Z","timestamp":1768022111000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["BayesAdapter: Enhanced Uncertainty Estimation in CLIP Few-Shot Adaptation"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2793-0083","authenticated-orcid":false,"given":"Pablo","family":"Morales-\u00c1lvarez","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stergios","family":"Christodoulidis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maria","family":"Vakalopoulou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pablo","family":"Piantanida","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jose","family":"Dolz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,1,10]]},"reference":[{"key":"2630_CR1","unstructured":"Baumann, A., Li, R., Klasson, M., Mentu, S., Karthik, S., Akata, Z., & Trapp, M. (2024). Post-hoc probabilistic vision-language models. arXiv:2412.06014"},{"issue":"518","key":"2630_CR2","doi-asserted-by":"publisher","first-page":"859","DOI":"10.1080\/01621459.2017.1285773","volume":"112","author":"DM Blei","year":"2017","unstructured":"Blei, D. M., Kucukelbir, A., & McAuliffe, J. D. (2017). Variational inference: A review for statisticians. Journal of the American statistical Association, 112(518), 859\u2013877.","journal-title":"Journal of the American statistical Association"},{"key":"2630_CR3","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., & Van\u00a0Gool, L. (2014). Food-101 \u2013 mining discriminative components with random forests. European conference on computer vision (eccv).","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"2630_CR4","unstructured":"Botev, A., Ritter, H., & Barber, D. (2017). Practical gauss-newton optimisation for deep learning. Proceedings of the 34th international conference on machine learning - volume 70 (p.557\u2013565). JMLR.org."},{"key":"2630_CR5","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., & Joulin, A. (2021). Emerging properties in self-supervised vision transformers. Proceedings of the ieee\/cvf international conference on computer vision (pp. 9650\u20139660).","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"2630_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2024.104115","volume":"331","author":"FM Castro-Mac\u00edas","year":"2024","unstructured":"Castro-Mac\u00edas, F. M., Morales-\u00c1lvarez, P., Wu, Y., Molina, R., & Katsaggelos, A. K. (2024). Hyperbolic secant representation of the logistic function: Application to probabilistic multiple instance learning for ct intracranial hemorrhage detection. Artificial Intelligence, 331, Article 104115.","journal-title":"Artificial Intelligence"},{"key":"2630_CR7","doi-asserted-by":"crossref","unstructured":"Chen, X., Xie, S., & He, K. (2021). An empirical study of training self-supervised vision transformers. Proceedings of the ieee\/cvf international conference on computer vision (pp. 9640\u20139649).","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"2630_CR8","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., & Vedaldi, A. (2014). Describing textures in the wild. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (p.3606-3613).","DOI":"10.1109\/CVPR.2014.461"},{"key":"2630_CR9","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L-J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (pp. 248\u2013255).","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2630_CR10","doi-asserted-by":"crossref","unstructured":"Derakhshani, M.M., Sanchez, E., Bulat, A., da Costa, V.G.T., Snoek, C.G., Tzimiropoulos, & G., Martinez, B. (2023). Bayesian prompt learning for image-language model generalization. Proceedings of the ieee\/cvf international conference on computer vision (iccv) (p.15237-15246).","DOI":"10.1109\/ICCV51070.2023.01398"},{"key":"2630_CR11","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., & Houlsby, N. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. International Conference on Learning Representations (ICLR)"},{"key":"2630_CR12","unstructured":"Fan, L., Krishnan, D., Isola, P., Katabi, D., & Tian, Y. (2024). Improving clip training with language rewrites. Advances in Neural Information Processing Systems, 36"},{"key":"2630_CR13","doi-asserted-by":"crossref","unstructured":"Fei-Fei, L., Fergus, R., & Perona, P. (2004). Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition worskshops (cvprw) (pp. 178-178).","DOI":"10.1109\/CVPR.2004.383"},{"key":"2630_CR14","doi-asserted-by":"crossref","unstructured":"Fu, H., Patel, N., Krishnamurthy, P., & Khorrami, F. (2024). Clipscope: Enhancing zero-shot ood detection with bayesian scoring. arXiv:2405.14737","DOI":"10.1109\/WACV61041.2025.00522"},{"key":"2630_CR15","doi-asserted-by":"crossref","unstructured":"Futami, F., & Fujisawa, M. (2024). Information-theoretic generalization analysis for expected calibration error. In Globerson, A. et\u00a0al. (Eds.), Advances in neural information processing systems (Vol. 37, pp. 84246\u201384297). Curran Associates, Inc.","DOI":"10.52202\/079017-2677"},{"key":"2630_CR16","unstructured":"Gal, Y., & Ghahramani, Z. (2016). Dropout as a bayesian approximation: Representing model uncertainty in deep learning. International conference on machine learning (pp. 1050\u20131059)."},{"key":"2630_CR17","doi-asserted-by":"crossref","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., & Qiao, Y. (2024). Clip-adapter: Better vision-language models with feature adapters. International Journal of Computer Vision (IJCV)","DOI":"10.1007\/s11263-023-01891-x"},{"key":"2630_CR18","unstructured":"Geifman, Y., & El-Yaniv, R. (2017). Selective classification for deep neural networks. In Guyon, I. et\u00a0al. (Eds.), Advances in neural information processing systems (Vol.\u00a030). Curran Associates, Inc."},{"key":"2630_CR19","unstructured":"Gomes, E.D.C., Romanelli, M., Pichler, G., & Piantanida, P. (2024). A data-driven measure of relative uncertainty for misclassification detection. The twelfth international conference on learning representations."},{"key":"2630_CR20","unstructured":"Guo, C., Pleiss, G., Sun, Y., & Weinberger, K.Q. (2017). On calibration of modern neural networks. In Precup, D., & Teh, Y.W. (Eds.), Proceedings of the 34th international conference on machine learning (Vol.\u00a070, pp. 1321\u20131330). PMLR."},{"key":"2630_CR21","doi-asserted-by":"crossref","unstructured":"Hantao Yao, C.X., Rui Zhang (2023). Visual-language prompt tuning with knowledge-guided context optimization. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr).","DOI":"10.1109\/CVPR52729.2023.00653"},{"issue":"9","key":"2630_CR22","doi-asserted-by":"publisher","first-page":"11602","DOI":"10.1109\/TITS.2024.3381175","volume":"25","author":"MZ Hasan","year":"2024","unstructured":"Hasan, M. Z., Chen, J., Wang, J., Rahman, M. S., Joshi, A., Velipasalar, S., & Sarkar, S. (2024). Vision-language models can identify distracted driver behavior from naturalistic videos. IEEE Transactions on Intelligent Transportation Systems, 25(9), 11602\u201311616. https:\/\/doi.org\/10.1109\/TITS.2024.3381175","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"2630_CR23","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2630_CR24","doi-asserted-by":"crossref","unstructured":"Helber, P., Bischke, B., Dengel, A., & Borth, D. (2018). Introducing eurosat: A novel dataset and deep learning benchmark for land use and land cover classification. Ieee international geoscience and remote sensing symposium (igarss) (pp. 3606-3613).","DOI":"10.1109\/IGARSS.2018.8519248"},{"key":"2630_CR25","unstructured":"Hendrycks, D., & Gimpel, K. (2017). A baseline for detecting misclassified and out-of-distribution examples in neural networks. International conference on learning representations."},{"issue":"1","key":"2630_CR26","first-page":"1303","volume":"14","author":"MD Hoffman","year":"2013","unstructured":"Hoffman, M. D., Blei, D. M., Wang, C., & Paisley, J. (2013). Stochastic variational inference. Journal of Machine Learning Research, 14(1), 1303\u20131347.","journal-title":"Journal of Machine Learning Research"},{"key":"2630_CR27","unstructured":"Huang, C- W., Tan, S., Lacoste, A., & Courville, A.C. (2018). Improving explorability in variational inference with annealed variational objectives. Advances in Neural Information Processing Systems, 31"},{"key":"2630_CR28","doi-asserted-by":"crossref","unstructured":"Huang, Y., Shakeri, F., Dolz, J., Boudiaf, M., Bahig, H., & Ayed, I.B. (2024). Lp++: A surprisingly strong linear probe for few-shot clip. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (pp. 23773\u201323782).","DOI":"10.1109\/CVPR52733.2024.02244"},{"key":"2630_CR29","unstructured":"Izmailov, P., Vikram, S., Hoffman, M.D., & Wilson, A.G.G. (2021). What are bayesian neural network posteriors really like? International conference on machine learning (pp. 4629\u20134640)."},{"key":"2630_CR30","first-page":"18211","volume":"35","author":"S Kapoor","year":"2022","unstructured":"Kapoor, S., Maddox, W. J., Izmailov, P., & Wilson, A. G. (2022). On uncertainty, tempering, and data augmentation in bayesian classification. Advances in Neural Information Processing Systems, 35, 18211\u201318225.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2630_CR31","doi-asserted-by":"crossref","unstructured":"Khan, Z., & Fu, Y. (2024). Consistency and uncertainty: Identifying unreliable responses from black-box vision-language models for selective visual question answering. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 10854\u201310863).","DOI":"10.1109\/CVPR52733.2024.01032"},{"key":"2630_CR32","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., & Khan, F.S. (2023). Maple: Multi-modal prompt learning. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (pp. 19113\u201319122).","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"2630_CR33","first-page":"21696","volume":"34","author":"D Kingma","year":"2021","unstructured":"Kingma, D., Salimans, T., Poole, B., & Ho, J. (2021). Variational diffusion models. Advances in Neural Information Processing Systems, 34, 21696\u201321707.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2630_CR34","unstructured":"Kingma, D.P. (2013). Auto-encoding variational bayes. arXiv:1312.6114"},{"key":"2630_CR35","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., & Fei-Fei, L. (2012). 3d object representations for fine-grained categorization. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (p.3498\u20133505).","DOI":"10.1109\/ICCVW.2013.77"},{"key":"2630_CR36","doi-asserted-by":"crossref","unstructured":"Lin, Z., Yu, S., Kuang, Z., Pathak, D., Ramanan, D. (2023). Multimodality helps unimodality: Cross-modal few-shot learning with multimodal models. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr).","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"2630_CR37","doi-asserted-by":"crossref","unstructured":"Liu, J., Zhang, Y., Chen, J-N., Xiao, J., Lu, Y., Landman, B. A., & Zhou, Z. (2023). Clip-driven universal model for organ segmentation and tumor detection. Proceedings of the ieee\/cvf international conference on computer vision (pp. 21152\u201321164).","DOI":"10.1109\/ICCV51070.2023.01934"},{"key":"2630_CR38","unstructured":"Liu, W., Wang, X., Owens, J., & Li, Y. (2020). Energy-based out-of-distribution detection. In Larochelle, H., Ranzato, M., Hadsell, R., Balcan,M., & Lin, H. (Eds.), Advances in neural information processing systems (Vol.\u00a033, pp. 21464\u201321475). Curran Associates, Inc."},{"key":"2630_CR39","unstructured":"MacKay, D. (1991). Bayesian model comparison and backprop nets. In Moody, J., Hanson, S., & Lippmann, R. (Eds.), Advances in neural information processing systems (Vol. 4). Morgan-Kaufmann."},{"key":"2630_CR40","unstructured":"Maji, S., Kannala, J., Rahtu, E., Blaschko, M., & Vedaldi, A. (2013). Fine-grained visual classification of aircraft. Arxiv preprint."},{"key":"2630_CR41","unstructured":"Martens, J., & Grosse, R. (2015). Optimizing neural networks with kronecker-factored approximate curvature. In Bach, F., & Blei, D. (Eds.), Proceedings of the 32nd international conference on machine learning (Vol. 37, pp. 2408\u20132417). Lille, France: PMLR. https:\/\/proceedings.mlr.press\/v37\/martens15.html"},{"key":"2630_CR42","doi-asserted-by":"crossref","unstructured":"Miao, Y., Lei, Y., Zhou, F., & Deng, Z. (2024). Bayesian exploration of pre-trained models for low-shot image classification. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 23849\u201323859).","DOI":"10.1109\/CVPR52733.2024.02251"},{"key":"2630_CR43","unstructured":"Morales-Alvarez, P., Hern\u00e1ndez-Lobato, D., Molina, R., & Hern\u00e1ndez-Lobato, J.M. (2020). Activation-level uncertainty in deep neural networks. International conference on learning representations."},{"issue":"3","key":"2630_CR44","doi-asserted-by":"publisher","first-page":"1534","DOI":"10.1109\/TPAMI.2020.3025390","volume":"44","author":"P Morales-\u00c1lvarez","year":"2022","unstructured":"Morales-\u00c1lvarez, P., Ruiz, P., Coughlin, S., Molina, R., & Katsaggelos, A. K. (2022). Scalable variational gaussian processes for crowdsourcing: Glitch detection in ligo. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(3), 1534\u20131551.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2630_CR45","doi-asserted-by":"crossref","unstructured":"Murugesan, B., Silva-Rodriguez, J., Ayed, I.B., & Dolz, J. (2024). Robust calibration of large vision-language adapters. European conference on computer vision (eccv).","DOI":"10.1007\/978-3-031-72691-0_9"},{"key":"2630_CR46","doi-asserted-by":"crossref","unstructured":"Nilsback, M-E., & Zisserman, A. (2008). Automated flower classification over a large number of classes. Indian conference on computer vision, graphics and image processing.","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"2630_CR47","unstructured":"Nixon, J., Dusenberry, M.W., Zhang, L., Jerfel, G., & Tran, D. (2019). Measuring calibration in deep learning. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) workshops."},{"key":"2630_CR48","unstructured":"Oh, C., Kim, M., Lim, H., Park, J., Jeong, E., Cheng, Z-. Q., & Song, K. (2024). Towards calibrated robust fine-tuning of vision-language models. Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"2630_CR49","unstructured":"Papamarkou, T., Skoularidou, M., Palla, K., Aitchison, L., Arbel, J., Dunson, D., & Zhang, R. (2024). Position: Bayesian deep learning is needed in the age of large-scale AI. In Salakhutdinov, R. et\u00a0al. (Eds.), Proceedings of the 41st international conference on machine learning (Vol.\u00a0235, pp. 39556\u201339586). PMLR."},{"key":"2630_CR50","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., & Jawahar, C. (2012). Cats and dogs. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (p.3498\u20133505).","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"2630_CR51","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., & Agarwal, S., et al. (2021). Learning transferable visual models from natural language supervision. International conference on machine learning (icml) (pp. 8748\u20138763)."},{"key":"2630_CR52","unstructured":"Ritter, H., Botev, A., & Barber, D. (2018). A scalable laplace approximation for neural networks. International conference on learning representations. Retrieved from https:\/\/openreview.net\/forum?id=Skdvd2xAZ."},{"key":"2630_CR53","doi-asserted-by":"crossref","unstructured":"Sain, A., Bhunia, A.K., Chowdhury, P.N., Koley, S., Xiang, T., & Song, Y-.Z. (2023). Clip for all things zero-shot sketch-based image retrieval, fine-grained or not. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 2765\u20132775).","DOI":"10.1109\/CVPR52729.2023.00271"},{"key":"2630_CR54","doi-asserted-by":"crossref","unstructured":"Silva-Rodriguez, J., Hajimiri, S., Ben\u00a0Ayed, I., & Dolz, J. (2024). A closer look at the few-shot adaptation of large vision-language models. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 23681\u201323690).","DOI":"10.1109\/CVPR52733.2024.02235"},{"key":"2630_CR55","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). Ucf101: A dataset of 101 human actions classes from videos in the wild. Arxiv preprint."},{"key":"2630_CR56","unstructured":"Tu, W., Deng, W., Campbell, D., Gould, S., & Gedeon, T. (2024). An empirical study into what matters for calibrating vision-language models. International conference on machine learning (icml)."},{"key":"2630_CR57","doi-asserted-by":"crossref","unstructured":"Upadhyay, U., Karthik, S., Mancini, M., & Akata, Z. (2023). Probvlm: Probabilistic adapter for frozen vison-language models. Proceedings of the ieee\/cvf international conference on computer vision (pp. 1899\u20131910).","DOI":"10.1109\/ICCV51070.2023.00182"},{"key":"2630_CR58","doi-asserted-by":"crossref","unstructured":"Wang, F., Mei, J., & Yuille, A. (2025). Sclip: Rethinking self-attention for dense vision-language inference. European conference on computer vision (pp. 315\u2013332).","DOI":"10.1007\/978-3-031-72664-4_18"},{"key":"2630_CR59","unstructured":"Wang, S., Wang, J., Wang, G., Zhang, B., Zhou, K., & Wei, H. (2024). Open-vocabulary calibration for fine-tuned clip. Forty-first international conference on machine learning."},{"key":"2630_CR60","doi-asserted-by":"crossref","unstructured":"Wang, S., Yang, C-.H.H., Wu, J., & Zhang, C. (2024). Bayesian example selection improves in-context learning for speech, text, and visual modalities. arXiv preprint arXiv:2404.14716","DOI":"10.18653\/v1\/2024.emnlp-main.1158"},{"key":"2630_CR61","unstructured":"Wilson, A.G., & Izmailov, P. (2020). Bayesian deep learning and a probabilistic perspective of generalization. In Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., & Lin, H. (Eds.), Advances in neural information processing systems (Vol. 33, pp. 4697\u20134708). Curran Associates, Inc."},{"key":"2630_CR62","unstructured":"Wu, Y-.C., Lyu, S-.H., Shang, H., Wang, X., & Qian, C. (2024). Confidence-aware contrastive learning for selective classification. International conference on machine learning."},{"key":"2630_CR63","unstructured":"Xia, G., Laurent, O., Franchi, G., & Bouganis, C-. S. (2024). Understanding why label smoothing degrades selective classification and how to fix it. arXiv preprint arXiv:2403.14715"},{"key":"2630_CR64","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (p.3485-3492).","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"2630_CR65","unstructured":"Yoon, H.S., Yoon, E., Tee, J.T.J., Hasegawa-Johnson, M.A., Li, Y., & Yoo, C.D. (2024). C-tpt: Calibrated test-time prompt tuning for vision-language models via text feature dispersion. The twelfth international conference on learning representations."},{"key":"2630_CR66","doi-asserted-by":"crossref","unstructured":"Yu, T., Lu, Z., Jin, X., Chen, Z., & Wang, X. (2023). Task residual for tuning vision-language models. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (pp. 10899\u201310909).","DOI":"10.1109\/CVPR52729.2023.01049"},{"issue":"8","key":"2630_CR67","doi-asserted-by":"publisher","first-page":"2008","DOI":"10.1109\/TPAMI.2018.2889774","volume":"41","author":"C Zhang","year":"2018","unstructured":"Zhang, C., B\u00fctepage, J., Kjellstr\u00f6m, H., & Mandt, S. (2018). Advances in variational inference. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(8), 2008\u20132026.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2630_CR68","unstructured":"Zhang, R., Fang, R., Zhang, W., Gao, P., Li, K., Dai, J., & Li, H. (2022). Tip-adapter: Training-free clip-adapter for better vision-language modeling. European conference on computer vision (eccv) (p.1-19)."},{"key":"2630_CR69","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022a). Conditional prompt learning for vision-language models. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (pp. 16816\u201316825).","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"2630_CR70","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., & Liu, Z. (2022b). Learning to prompt for vision-language models. International Journal of Computer Vision (IJCV)","DOI":"10.1007\/s11263-022-01653-1"},{"key":"2630_CR71","doi-asserted-by":"crossref","unstructured":"Zhu, B., Niu, Y., Han, Y., Wu, Y., & Zhang, H. (2023). Prompt-aligned gradient for prompt tuning. Proceedings of the ieee\/cvf conference on computer vision and pattern recognition (cvpr) (pp. 15659\u201315669).","DOI":"10.1109\/ICCV51070.2023.01435"},{"key":"2630_CR72","doi-asserted-by":"crossref","unstructured":"Zhu, L., Wang, X., Zhou, C., & Ye, N. (2023). Bayesian cross-modal alignment learning for few-shot out-of-distribution generalization. Proceedings of the aaai conference on artificial intelligence (Vol.\u00a037, pp. 11461\u201311469).","DOI":"10.1609\/aaai.v37i9.26355"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02630-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02630-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02630-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T15:21:36Z","timestamp":1771341696000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02630-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,10]]},"references-count":72,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2630"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02630-0","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,10]]},"assertion":[{"value":"9 April 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"Our code will be publicly available and is submitted in the supplementary material.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}],"article-number":"51"}}