{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T21:04:19Z","timestamp":1784149459487,"version":"3.55.0"},"reference-count":64,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62233003"],"award-info":[{"award-number":["62233003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U22B2040"],"award-info":[{"award-number":["U22B2040"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116568","type":"journal-article","created":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T15:28:03Z","timestamp":1782919683000},"page":"116568","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Hierarchical cross-modal prompt learning with LLM-guided alignment for Vision-Language Models"],"prefix":"10.1016","volume":"350","author":[{"given":"Wenyu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5810-7631","authenticated-orcid":false,"given":"Fuxiang","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shi","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116568_b1","unstructured":"A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, et al., Learning transferable visual models from natural language supervision, in: International Conference on Machine Learning, 2021, pp. 8748\u20138763."},{"key":"10.1016\/j.knosys.2026.116568_b2","unstructured":"J. Li, D. Li, C. Xiong, S. Hoi, BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation, in: Proceedings of the 39th International Conference on Machine Learning, Vol. 162, 2022, pp. 12888\u201312900."},{"key":"10.1016\/j.knosys.2026.116568_b3","unstructured":"J. Li, D. Li, S. Savarese, S. Hoi, BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models, in: Proceedings of the 40th International Conference on Machine Learning, Vol. 202, 2023, pp. 19730\u201319742."},{"key":"10.1016\/j.knosys.2026.116568_b4","doi-asserted-by":"crossref","unstructured":"Y. Yang, A. Panagopoulou, S. Zhou, D. Jin, C. Callison-Burch, M. Yatskar, Language in a Bottle: Language Model Guided Concept Bottlenecks for Interpretable Image Classification, in: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 19187\u201319197.","DOI":"10.1109\/CVPR52729.2023.01839"},{"key":"10.1016\/j.knosys.2026.116568_b5","series-title":"Visualbert: A simple and performant baseline for vision and language","author":"Li","year":"2019"},{"key":"10.1016\/j.knosys.2026.116568_b6","series-title":"A good prompt is worth millions of parameters: Low-resource prompt-based learning for vision-language models","author":"Jin","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b7","series-title":"CPT: Colorful prompt tuning for pre-trained vision-language models","author":"Yao","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b8","series-title":"Exploring visual prompts for adapting large-scale models","author":"Bahng","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b9","series-title":"Advances in Neural Information Processing Systems","first-page":"13834","article-title":"Toalign: Task-oriented alignment for unsupervised domain adaptation","volume":"Vol. 34","author":"Wei","year":"2021"},{"key":"10.1016\/j.knosys.2026.116568_b10","doi-asserted-by":"crossref","unstructured":"M.F. Naeem, M.G.Z. Ali Khan, Y. Xian, M.Z. Afzal, D. Stricker, L. Van Gool, F. Tombari, I2MVFormer: Large Language Model Generated Multi-View Document Supervision for Zero-Shot Image Classification, in: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 15169\u201315179.","DOI":"10.1109\/CVPR52729.2023.01456"},{"key":"10.1016\/j.knosys.2026.116568_b11","unstructured":"A. Ramesh, M. Pavlov, G. Goh, S. Gray, C. Voss, A. Radford, M. Chen, I. Sutskever, Zero-Shot Text-to-Image Generation, in: Proceedings of the 38th International Conference on Machine Learning, Vol. 139, 2021, pp. 8821\u20138831."},{"key":"10.1016\/j.knosys.2026.116568_b12","series-title":"Visual classification via description from large language models","author":"Menon","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b13","series-title":"Advances in Neural Information Processing Systems","first-page":"23049","article-title":"Towards understanding the mixture-of-experts layer in deep learning","volume":"Vol. 35","author":"Chen","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b14","series-title":"Computer Vision \u2013 ECCV 2022","first-page":"105","article-title":"Prompting visual-language models for efficient video understanding","author":"Ju","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b15","doi-asserted-by":"crossref","unstructured":"Y. Rao, W. Zhao, G. Chen, Y. Tang, Z. Zhu, G. Huang, J. Zhou, J. Lu, DenseCLIP: Language-Guided Dense Prediction with Context-Aware Prompting, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 18061\u201318070.","DOI":"10.1109\/CVPR52688.2022.01755"},{"issue":"7","key":"10.1016\/j.knosys.2026.116568_b16","first-page":"7377","article-title":"Concept-guided prompt learning for generalization in vision-language models","volume":"38","author":"Zhang","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"6","key":"10.1016\/j.knosys.2026.116568_b17","first-page":"5749","article-title":"Learning hierarchical prompt with structured linguistic knowledge for vision-language models","volume":"38","author":"Wang","year":"2024","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.116568_b18","doi-asserted-by":"crossref","unstructured":"Y. Lu, J. Liu, Y. Zhang, Y. Liu, X. Tian, Prompt Distribution Learning, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 5196\u20135205.","DOI":"10.1109\/CVPR52688.2022.00514"},{"key":"10.1016\/j.knosys.2026.116568_b19","series-title":"Advances in Neural Information Processing Systems","first-page":"33781","article-title":"Bridging the gap between object and image-level representations for open-vocabulary detection","volume":"Vol. 35","author":"Bangalath","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b20","doi-asserted-by":"crossref","unstructured":"B. Lester, R. Al-Rfou, N. Constant, The Power of Scale for Parameter-Efficient Prompt Tuning, in: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, 2021, pp. 3045\u20133059.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"10.1016\/j.knosys.2026.116568_b21","doi-asserted-by":"crossref","unstructured":"E. Cho, J. Kim, H.J. Kim, Distribution-Aware Prompt Tuning for Vision-Language Models, in: 2023 IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 21947\u201321956.","DOI":"10.1109\/ICCV51070.2023.02011"},{"key":"10.1016\/j.knosys.2026.116568_b22","doi-asserted-by":"crossref","unstructured":"B. Zhu, Y. Niu, Y. Han, Y. Wu, H. Zhang, Prompt-aligned Gradient for Prompt Tuning, in: 2023 IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 15613\u201315623.","DOI":"10.1109\/ICCV51070.2023.01435"},{"key":"10.1016\/j.knosys.2026.116568_b23","doi-asserted-by":"crossref","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","article-title":"Learning to prompt for vision-language models","volume":"130","author":"Zhou","year":"2022","journal-title":"Int. J. Comput. Vis. (IJCV)"},{"key":"10.1016\/j.knosys.2026.116568_b24","doi-asserted-by":"crossref","unstructured":"K. Zhou, J. Yang, C.C. Loy, Z. Liu, Conditional Prompt Learning for Vision-Language Models, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 16795\u201316804.","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"10.1016\/j.knosys.2026.116568_b25","series-title":"Computer Vision \u2013 ECCV 2022","first-page":"709","article-title":"Visual prompt tuning","author":"Jia","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b26","doi-asserted-by":"crossref","unstructured":"M.U. Khattak, H. Rasheed, M. Maaz, S. Khan, F.S. Khan, MaPLe: Multi-modal Prompt Learning, in: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 19113\u201319122.","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"10.1016\/j.knosys.2026.116568_b27","doi-asserted-by":"crossref","first-page":"3469","DOI":"10.1109\/TMM.2023.3311646","article-title":"Sgva-CLIP: Semantic-guided visual adapting of vision-language models for few-shot image classification","volume":"26","author":"Peng","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.knosys.2026.116568_b28","doi-asserted-by":"crossref","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","article-title":"CLIP-adapter: Better vision-language models with feature adapters","volume":"132","author":"Gao","year":"2024","journal-title":"Int. J. Comput. Vis. (IJCV)"},{"key":"10.1016\/j.knosys.2026.116568_b29","doi-asserted-by":"crossref","first-page":"392","DOI":"10.1007\/s11263-023-01876-w","article-title":"Transferring vision-language models for visual recognition: A classifier perspective","volume":"132","author":"Wu","year":"2024","journal-title":"Int. J. Comput. Vis. (IJCV)"},{"key":"10.1016\/j.knosys.2026.116568_b30","doi-asserted-by":"crossref","unstructured":"W.-H. Li, X. Liu, H. Bilen, Cross-domain Few-shot Learning with Task-specific Adapters, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 7151\u20137160.","DOI":"10.1109\/CVPR52688.2022.00702"},{"key":"10.1016\/j.knosys.2026.116568_b31","unstructured":"C. Jia, Y. Yang, Y. Xia, Y.-T. Chen, Z. Parekh, H. Pham, Q. Le, Y.-H. Sung, Z. Li, T. Duerig, Scaling up visual and vision-language representation learning with noisy text supervision, in: Proceedings of the 38th International Conference on Machine Learning, 2021, pp. 4904\u20134916."},{"key":"10.1016\/j.knosys.2026.116568_b32","series-title":"FILIP: Fine-grained interactive language-image pre-training","author":"Yao","year":"2021"},{"key":"10.1016\/j.knosys.2026.116568_b33","doi-asserted-by":"crossref","unstructured":"Y. Du, F. Wei, Z. Zhang, M. Shi, Y. Gao, G. Li, Learning to Prompt for Open-Vocabulary Object Detection with Vision-Language Model, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 14064\u201314073.","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"10.1016\/j.knosys.2026.116568_b34","series-title":"Computer Vision \u2013 ECCV 2022","first-page":"701","article-title":"PromptDet: Towards open-vocabulary detection using uncurated images","author":"Feng","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b35","unstructured":"S. Santurkar, Y. Dubois, R. Taori, P. Liang, T. Hashimoto, Is a Caption Worth a Thousand Images? A Study on Representation Learning, in: Proceedings of the International Conference on Learning Representations, 2023."},{"key":"10.1016\/j.knosys.2026.116568_b36","series-title":"Computer Vision \u2013 ECCV 2020","first-page":"266","article-title":"Rethinking few-shot image classification: A good embedding is all you need?","author":"Tian","year":"2020"},{"key":"10.1016\/j.knosys.2026.116568_b37","doi-asserted-by":"crossref","unstructured":"X. Hu, Z. Gan, J. Wang, Z. Yang, Z. Liu, Y. Lu, L. Wang, Scaling Up Vision-Language Pretraining for Image Captioning, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 17959\u201317968.","DOI":"10.1109\/CVPR52688.2022.01745"},{"key":"10.1016\/j.knosys.2026.116568_b38","series-title":"Advances in Neural Information Processing Systems","first-page":"23716","article-title":"Flamingo: a visual language model for few-shot learning","volume":"Vol. 35","author":"Alayrac","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b39","doi-asserted-by":"crossref","unstructured":"X.L. Li, P. Liang, Prefix-Tuning: Optimizing Continuous Prompts for Generation, in: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), 2021, pp. 4582\u20134597.","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"10.1016\/j.knosys.2026.116568_b40","series-title":"Computer Vision \u2013 ECCV 2022","first-page":"493","article-title":"Tip-adapter: Training-free adaption of CLIP for few-shot classification","author":"Zhang","year":"2022"},{"key":"10.1016\/j.knosys.2026.116568_b41","series-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"10.1016\/j.knosys.2026.116568_b42","series-title":"Gpt-4: A review on advancements and opportunities in natural language processing","author":"Baktash","year":"2023"},{"key":"10.1016\/j.knosys.2026.116568_b43","series-title":"Advances in Neural Information Processing Systems","first-page":"34892","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023"},{"key":"10.1016\/j.knosys.2026.116568_b44","series-title":"Advances in Neural Information Processing Systems","first-page":"1877","article-title":"Language models are few-shot learners","volume":"Vol. 33","author":"Brown","year":"2020"},{"key":"10.1016\/j.knosys.2026.116568_b45","doi-asserted-by":"crossref","unstructured":"S. Pratt, I. Covert, R. Liu, A. Farhadi, What does a platypus look like? Generating customized prompts for zero-shot image classification, in: 2023 IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 15645\u201315655.","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"10.1016\/j.knosys.2026.116568_b46","doi-asserted-by":"crossref","unstructured":"H. Yao, R. Zhang, C. Xu, TCP: Textual-Based Class-Aware Prompt Tuning for Visual-Language Model, in: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 23438\u201323448.","DOI":"10.1109\/CVPR52733.2024.02212"},{"key":"10.1016\/j.knosys.2026.116568_b47","doi-asserted-by":"crossref","unstructured":"J. Deng, W. Dong, R. Socher, L.-J. Li, K. Li, L. Fei-Fei, ImageNet: A large-scale hierarchical image database, in: 2009 IEEE Conference on Computer Vision and Pattern Recognition, 2009, pp. 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"10.1016\/j.knosys.2026.116568_b48","doi-asserted-by":"crossref","unstructured":"L. Fei-Fei, R. Fergus, P. Perona, Learning Generative Visual Models from Few Training Examples: An Incremental Bayesian Approach Tested on 101 Object Categories, in: 2004 Conference on Computer Vision and Pattern Recognition Workshop, 2004, pp. 178\u2013178.","DOI":"10.1109\/CVPR.2004.383"},{"key":"10.1016\/j.knosys.2026.116568_b49","doi-asserted-by":"crossref","unstructured":"O.M. Parkhi, A. Vedaldi, A. Zisserman, C.V. Jawahar, Cats and dogs, in: 2012 IEEE Conference on Computer Vision and Pattern Recognition, 2012, pp. 3498\u20133505.","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"10.1016\/j.knosys.2026.116568_b50","doi-asserted-by":"crossref","unstructured":"J. Krause, M. Stark, J. Deng, L. Fei-Fei, 3D Object Representations for Fine-Grained Categorization, in: 2013 IEEE International Conference on Computer Vision Workshops, 2013, pp. 554\u2013561.","DOI":"10.1109\/ICCVW.2013.77"},{"key":"10.1016\/j.knosys.2026.116568_b51","doi-asserted-by":"crossref","unstructured":"M.-E. Nilsback, A. Zisserman, Automated Flower Classification over a Large Number of Classes, in: 2008 Sixth Indian Conference on Computer Vision, Graphics & Image Processing, 2008, pp. 722\u2013729.","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"10.1016\/j.knosys.2026.116568_b52","series-title":"Computer Vision \u2013 ECCV 2014","first-page":"446","article-title":"Food-101 \u2013 mining discriminative components with random forests","author":"Bossard","year":"2014"},{"key":"10.1016\/j.knosys.2026.116568_b53","series-title":"Fine-grained visual classification of aircraft","author":"Maji","year":"2013"},{"key":"10.1016\/j.knosys.2026.116568_b54","doi-asserted-by":"crossref","unstructured":"J. Xiao, J. Hays, K.A. Ehinger, A. Oliva, A. Torralba, SUN database: Large-scale scene recognition from abbey to zoo, in: 2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition, 2010, pp. 3485\u20133492.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"10.1016\/j.knosys.2026.116568_b55","doi-asserted-by":"crossref","unstructured":"M. Cimpoi, S. Maji, I. Kokkinos, S. Mohamed, A. Vedaldi, Describing Textures in the Wild, in: 2014 IEEE Conference on Computer Vision and Pattern Recognition, 2014, pp. 3606\u20133613.","DOI":"10.1109\/CVPR.2014.461"},{"key":"10.1016\/j.knosys.2026.116568_b56","doi-asserted-by":"crossref","first-page":"2217","DOI":"10.1109\/JSTARS.2019.2918242","article-title":"EuroSAT: A novel dataset and deep learning benchmark for land use and land cover classification","volume":"12","author":"Helber","year":"2019","journal-title":"IEEE J. Sel. Top. Appl. Earth Obs. Remote. Sens."},{"key":"10.1016\/j.knosys.2026.116568_b57","series-title":"UCF101: A dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"10.1016\/j.knosys.2026.116568_b58","unstructured":"B. Recht, R. Roelofs, L. Schmidt, V. Shankar, Do ImageNet Classifiers Generalize to ImageNet?, in: Proceedings of the 36th International Conference on Machine Learning, 2019, pp. 5389\u20135400."},{"key":"10.1016\/j.knosys.2026.116568_b59","series-title":"Advances in Neural Information Processing Systems","article-title":"Learning robust global representations by penalizing local predictive power","volume":"Vol. 32","author":"Wang","year":"2019"},{"key":"10.1016\/j.knosys.2026.116568_b60","doi-asserted-by":"crossref","unstructured":"D. Hendrycks, K. Zhao, S. Basart, J. Steinhardt, D. Song, Natural Adversarial Examples, in: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2021, pp. 15257\u201315266.","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"10.1016\/j.knosys.2026.116568_b61","doi-asserted-by":"crossref","unstructured":"D. Hendrycks, S. Basart, N. Mu, S. Kadavath, F. Wang, E. Dorundo, R. Desai, T. Zhu, S. Parajuli, M. Guo, D. Song, J. Steinhardt, J. Gilmer, The Many Faces of Robustness: A Critical Analysis of Out-of-Distribution Generalization, in: 2021 IEEE\/CVF International Conference on Computer Vision, ICCV, 2021, pp. 8320\u20138329.","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"10.1016\/j.knosys.2026.116568_b62","doi-asserted-by":"crossref","unstructured":"H. Yao, R. Zhang, C. Xu, Visual-Language Prompt Tuning with Knowledge-Guided Context Optimization, in: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2023, pp. 6757\u20136767.","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"10.1016\/j.knosys.2026.116568_b63","doi-asserted-by":"crossref","first-page":"1348","DOI":"10.1109\/TIP.2024.3362062","article-title":"Learning domain invariant prompt for vision-language models","volume":"33","author":"Zhao","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.116568_b64","doi-asserted-by":"crossref","unstructured":"M.U. Khattak, S.T. Wasim, M. Naseer, S. Khan, M.-H. Yang, F.S. Khan, Self-regulating Prompts: Foundational Model Adaptation without Forgetting, in: 2023 IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 15144\u201315154.","DOI":"10.1109\/ICCV51070.2023.01394"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126012943?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126012943?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T20:06:07Z","timestamp":1784145967000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126012943"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":64,"alternative-id":["S0950705126012943"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116568","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Hierarchical cross-modal prompt learning with LLM-guided alignment for Vision-Language Models","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116568","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116568"}}