{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:51:10Z","timestamp":1782201070542,"version":"3.54.5"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00371-026-04549-0","type":"journal-article","created":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T06:56:17Z","timestamp":1780815377000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing vision\u2013language model calibration via rotation-enhanced orthogonality-constrained prompt tuning"],"prefix":"10.1007","volume":"42","author":[{"given":"Kai","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dandan","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,7]]},"reference":[{"issue":"4","key":"4549_CR1","doi-asserted-by":"publisher","first-page":"2245","DOI":"10.1109\/TPAMI.2024.3506283","volume":"47","author":"M Awais","year":"2025","unstructured":"Awais, M., Naseer, M., Khan, S., Anwer, R.M., Cholakkal, H., Shah, M., Yang, M.H., Khan, F.S.: Foundation models defining a new era in vision: a survey and outlook. IEEE Trans. Pattern Anal. Mach. Intell. 47(4), 2245\u20132264 (2025)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4549_CR2","doi-asserted-by":"crossref","unstructured":"Bossard, L., Guillaumin, M., Van Gool, L.: Food-101-mining discriminative components with random forests. In: European conference on computer vision, pp. 446\u2013461. Springer (2014)","DOI":"10.1007\/978-3-319-10599-4_29"},{"issue":"11","key":"4549_CR3","doi-asserted-by":"crossref","first-page":"13489","DOI":"10.1109\/TPAMI.2023.3289667","volume":"45","author":"Z Chen","year":"2023","unstructured":"Chen, Z., Qiu, G., Li, P., Zhu, L., Yang, X., Sheng, B.: Mngnas: distilling adaptive combination of multiple searched networks for one-shot neural architecture search. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 13489\u201313508 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4549_CR4","doi-asserted-by":"crossref","unstructured":"Cimpoi, M., Maji, S., Kokkinos, I., Mohamed, S., Vedaldi, A.: Describing textures in the wild. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3606\u20133613. (2014)","DOI":"10.1109\/CVPR.2014.461"},{"key":"4549_CR5","doi-asserted-by":"publisher","first-page":"49250","DOI":"10.52202\/075280-2142","volume":"36","author":"W Dai","year":"2023","unstructured":"Dai, W., Li, J., Li, D., Tiong, A., Zhao, J., Wang, W., Li, B., Fung, P.N., Hoi, S.: Instructblip: towards general-purpose vision-language models with instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 49250\u201349267 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4549_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: A large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition. pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4549_CR7","unstructured":"Fei-Fei, L., Fergus, R., Perona, P.: Learning generative visual models from few training examples: An incremental bayesian approach tested on 101 object categories. In: 2004 conference on computer vision and pattern recognition workshop. pp. 178\u2013178. IEEE (2004)"},{"key":"4549_CR8","unstructured":"Guo, C., Pleiss, G., Sun, Y., Weinberger, K.Q.: On calibration of modern neural networks. In: Proceedings of the 34th International Conference on Machine Learning (ICML), (2017)"},{"issue":"7","key":"4549_CR9","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","volume":"12","author":"P Helber","year":"2019","unstructured":"Helber, P., Bischke, B., Dengel, A., Borth, D.: Eurosat: a novel dataset and deep learning benchmark for land use and land cover classification. IEEE J. Sel. Topics Appl. Earth Observ. Remote Sensi. 12(7), 2217\u20132226 (2019)","journal-title":"IEEE J. Sel. Topics Appl. Earth Observ. Remote Sensi."},{"key":"4549_CR10","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.T., Parekh, Z., Pham, H., Le, Q., Sung, Y.H., Li, Z., Duerig, T.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"4549_CR11","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.C., Cardie, C., Belongie, S., Hariharan, B., Lim, S.N.:Visual prompt tuning. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 709\u2013727. (2022)","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"4549_CR12","unstructured":"Karandikar, A., Gu, A., Ma, T., R\u00e9, C.: Soft calibration objectives for neural networks. In: Advances in Neural Information Processing Systems (NeurIPS), (2021)"},{"key":"4549_CR13","doi-asserted-by":"crossref","unstructured":"Karmanov, A., Guan, D., Lu, S., El Saddik, A., Xing, E.: Efficient test-time adaptation of vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14162\u201314171. (2024)","DOI":"10.1109\/CVPR52733.2024.01343"},{"key":"4549_CR14","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y.: Segment anything. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 4015\u20134026. (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"4549_CR15","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., Fei-Fei, L.: 3d object representations for fine-grained categorization. In: Proceedings of the IEEE international conference on computer vision workshops, pp. 554\u2013561. (2013)","DOI":"10.1109\/ICCVW.2013.77"},{"key":"4549_CR16","unstructured":"Kumar, A., Liang, P., Ma, T.: Verified uncertainty calibration. In: Advances in Neural Information Processing Systems (NeurIPS), (2018)"},{"issue":"1\u20133","key":"4549_CR17","first-page":"203","volume":"102","author":"J Lei","year":"2018","unstructured":"Lei, J., Jelasity, M., Cs\u00e1ji, B.C.: Conformal prediction in supervised learning. Mach. Learn. 102(1\u20133), 203\u2013229 (2018)","journal-title":"Mach. Learn."},{"key":"4549_CR18","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International conference on machine learning. pp. 19730\u201319742. PMLR (2023)"},{"issue":"1","key":"4549_CR19","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1007\/s11263-024-02181-w","volume":"133","author":"J Liang","year":"2025","unstructured":"Liang, J., He, R., Tan, T.: A comprehensive survey on test-time adaptation under distribution shifts. Int. J. Comput. Vision 133(1), 31\u201364 (2025)","journal-title":"Int. J. Comput. Vision"},{"key":"4549_CR20","doi-asserted-by":"publisher","first-page":"34892","DOI":"10.52202\/075280-1516","volume":"36","author":"H Liu","year":"2023","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 34892\u201334916 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4549_CR21","doi-asserted-by":"crossref","unstructured":"L\u00fcddecke, T., Ecker, A.: Image segmentation using text and image prompts. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7086\u20137096. (2022)","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"4549_CR22","unstructured":"Lv, S.L., Chen, Y.Y., Zhou, Z., Li, Y.F., Guo, L.Z.: Contrast-aware calibration for fine-tuned clip: Leveraging image-text alignment, (2025). arXiv:2501.19060 arXiv preprint"},{"key":"4549_CR23","first-page":"2579","volume":"9","author":"L Maaten","year":"2008","unstructured":"Maaten, L., Hinton, G.: Visualizing data using t-sne. J. Mach. Learn. Res. 9, 2579\u20132605 (2008)","journal-title":"J. Mach. Learn. Res."},{"key":"4549_CR24","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., Vedaldi, A.: Fine-grained visual classification of aircraft, (2013). arXiv preprint arXiv:1306.5151"},{"key":"4549_CR25","doi-asserted-by":"crossref","unstructured":"Morales-\u00c1lvarez, P., Christodoulidis, S., Vakalopoulou, M., Piantanida, P., Dolz, J.: Bayesadapter: enhanced uncertainty estimation in clip few-shot adaptation, (2025)","DOI":"10.1007\/s11263-025-02630-0"},{"key":"4549_CR26","doi-asserted-by":"crossref","unstructured":"Murugesan, B., Silva-Rodr\u00edguez, J., Ayed, I.B., Dolz, J.: Robust calibration of large vision-language adapters. In: European Conference on Computer Vision, pp. 147\u2013165. Springer (2024)","DOI":"10.1007\/978-3-031-72691-0_9"},{"key":"4549_CR27","doi-asserted-by":"crossref","unstructured":"Naeini, M.P., Cooper, G., Hauskrecht, M.: Obtaining well calibrated probabilities using bayesian binning. In: Proceedings of the AAAI conference on artificial intelligence, vol. 29, (2015)","DOI":"10.1609\/aaai.v29i1.9602"},{"key":"4549_CR28","doi-asserted-by":"crossref","unstructured":"Nilsback, M.E., Zisserman, A.: Automated flower classification over a large number of classes. In: 2008 Sixth Indian conference on computer vision, graphics & image processing. pp. 722\u2013729. IEEE (2008)","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"4549_CR29","doi-asserted-by":"crossref","unstructured":"Oh, C., Lim, H., Kim, M., Han, D., Yun, S., Choo, J., Hauptmann, A.G., Cheng, Z.Q., Song, K.: Towards calibrated robust fine-tuning of vision-language models. In: The Thirty-eighth Annual Conference on Neural Information Processing Systems (2024), https:\/\/openreview.net\/forum?id=GnAfyR8AhC","DOI":"10.52202\/079017-0403"},{"key":"4549_CR30","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Vedaldi, A., Zisserman, A., Jawahar, C.: Cats and dogs. In: 2012 IEEE conference on computer vision and pattern recognition, pp. 3498\u20133505. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"4549_CR31","unstructured":"Platt, J.C.: Probabilistic outputs for support vector machines and comparisons to regularized likelihood methods. In: Advances in Large Margin Classifiers, MIT Press (1999)"},{"key":"4549_CR32","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J.: Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"4549_CR33","doi-asserted-by":"publisher","first-page":"66127","DOI":"10.52202\/075280-2888","volume":"36","author":"VV Ramaswamy","year":"2023","unstructured":"Ramaswamy, V.V., Lin, S.Y., Zhao, D., Adcock, A., van der Maaten, L., Ghadiyaram, D., Russakovsky, O.: Geode: a geographically diverse evaluation dataset for object recognition. Adv. Neural. Inf. Process. Syst. 36, 66127\u201366137 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4549_CR34","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. 1(2), 3 (2022). arXiv preprint arXiv:2204.06125"},{"key":"4549_CR35","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10684\u201310695. (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"4549_CR36","doi-asserted-by":"crossref","unstructured":"Sharifdeen, A., Munir, M.A., Baliah, S., Khan, S., Khan, M.H.: O-tpt: Orthogonality constraints for calibrating test-time prompt tuning in vision-language models. In: Proceedings of the Computer Vision and Pattern Recognition Conference. pp. 19942\u201319951 (2025)","DOI":"10.1109\/CVPR52734.2025.01857"},{"key":"4549_CR37","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 2556\u20132565 (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"4549_CR38","doi-asserted-by":"crossref","unstructured":"Shu, M., Liu, Z., Holtzman, A., Bansal, M., Yasunaga, M., Xie, S.: Test-time prompt tuning for zero-shot generalization in vision-language models. In: Advances in Neural Information Processing Systems (NeurIPS), (2022)","DOI":"10.52202\/068431-1038"},{"key":"4549_CR39","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: Ucf101: A dataset of 101 human actions classes from videos in the wild, (2012). arXiv preprint arXiv:1212.0402"},{"key":"4549_CR40","doi-asserted-by":"crossref","unstructured":"Thulasidasan, S., Chennupati, G., Bilmes, J.A., Bhattacharya, T., Michalak, S.: On mixup training: Improved calibration and predictive uncertainty for deep neural networks. In: Advances in Neural Information Processing Systems (NeurIPS), (2019)","DOI":"10.2172\/1525811"},{"key":"4549_CR41","unstructured":"Tu, W., Deng, W., Campbell, D., Gould, S., Gedeon, T.: An empirical study into what matters for calibrating vision-language models, (2024). arXiv preprint arXiv:2402.07417"},{"key":"4549_CR42","volume-title":"Algorithmic Learning in a Random World","author":"V Vovk","year":"2005","unstructured":"Vovk, V., Gammerman, A., Shafer, G.: Algorithmic Learning in a Random World. Springer (2005)"},{"key":"4549_CR43","doi-asserted-by":"crossref","unstructured":"Wang, C., Mahadevan, S.: Manifold alignment using procrustes analysis. In: Proceedings of the 25th international conference on Machine learning, pp. 1120\u20131127. (2008)","DOI":"10.1145\/1390156.1390297"},{"key":"4549_CR44","unstructured":"Wang, H., Liu, J., Liu, Q.: On improving neural network calibration with mixup training. In: Proceedings of the 40th International Conference on Machine Learning (ICML) (2023)"},{"key":"4549_CR45","doi-asserted-by":"crossref","unstructured":"Wang, Z., Wu, Z., Agarwal, D., Sun, J.: Medclip: Contrastive learning from unpaired medical images and text. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing. Conference on Empirical Methods in Natural Language Processing. vol.\u00a02022, p.\u00a03876 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"issue":"1","key":"4549_CR46","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s11263-014-0748-y","volume":"119","author":"J Xiao","year":"2016","unstructured":"Xiao, J., Ehinger, K.A., Hays, J., Torralba, A., Oliva, A.: Sun database: exploring a large collection of scene categories. Int. J. Comput. Vision 119(1), 3\u201322 (2016)","journal-title":"Int. J. Comput. Vision"},{"key":"4549_CR47","doi-asserted-by":"crossref","unstructured":"Xu, J., De Mello, S., Liu, S., Byeon, W., Breuel, T., Kautz, J., Wang, X.: Groupvit: Semantic segmentation emerges from text supervision. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 18134\u201318144. (2022)","DOI":"10.1109\/CVPR52688.2022.01760"},{"key":"4549_CR48","unstructured":"Yoon, H.S., Yoon, E., Hasegawa-Johnson, M., Yoo, C.D.: Esd: Efficient and scalable calibration via entropy-based surrogate loss. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"4549_CR49","unstructured":"Yoon, H.S., Yoon, E., Tee, J.T.J., Hasegawa-Johnson, M., Li, Y., Yoo, C.D.: C-tpt: Calibrated test-time prompt tuning for vision-language models via text feature dispersion. arXiv preprint arXiv:2403.14119 (2024)"},{"key":"4549_CR50","unstructured":"Zhang, H., Cisse, M., Dauphin, Y.N., Lopez-Paz, D.: mixup: Beyond empirical risk minimization. In: International Conference on Learning Representations (ICLR), (2018)"},{"key":"4549_CR51","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16816\u201316825. (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"4549_CR52","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"4549_CR53","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16597\u201316607. (2022)"},{"key":"4549_CR54","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: Minigpt-4: Enhancing vision-language understanding with advanced large language models, (2023). arXiv preprint arXiv:2304.10592"},{"issue":"6","key":"4549_CR55","doi-asserted-by":"publisher","first-page":"3891","DOI":"10.1109\/TSMC.2024.3374068","volume":"54","author":"J Zhu","year":"2024","unstructured":"Zhu, J., Chen, X., Hu, Q., Xiao, Y., Wang, B., Sheng, B., Chen, C.P.: Clustering environment aware learning for active domain adaptation. IEEE Trans. Syste. Man Cybern. Syst. 54(6), 3891\u20133904 (2024)","journal-title":"IEEE Trans. Syste. Man Cybern. Syst."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04549-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04549-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04549-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:40:47Z","timestamp":1782200447000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04549-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":55,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["4549"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04549-0","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"29 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"321"}}