{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T21:16:41Z","timestamp":1783027001321,"version":"3.54.6"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62566015"],"award-info":[{"award-number":["62566015"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U22A2099"],"award-info":[{"award-number":["U22A2099"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62336003"],"award-info":[{"award-number":["62336003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114161","type":"journal-article","created":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T16:23:16Z","timestamp":1780935796000},"page":"114161","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["A hierarchical evaluation framework for security and trustworthiness in Large Vision-Language Models"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1661-6590","authenticated-orcid":false,"given":"Xuan","family":"Feng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingfeng","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1593-1292","authenticated-orcid":false,"given":"Tianlong","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7262-4707","authenticated-orcid":false,"given":"Liang","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5609-163X","authenticated-orcid":false,"given":"Chenzhong","family":"Bin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114161_b1","doi-asserted-by":"crossref","unstructured":"G. Xu, P. Jin, Z. Wu, H. Li, Y. Song, L. Sun, L. Yuan, LLaVA-CoT: Let Vision Language Models Reason Step-by-Step, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 2087\u20132098.","DOI":"10.1109\/ICCV51701.2025.00202"},{"key":"10.1016\/j.patcog.2026.114161_b2","article-title":"Cross-scene visual context parsing with large vision-language model","author":"Zhang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114161_b3","doi-asserted-by":"crossref","unstructured":"Z. Shao, Z. Yu, M. Wang, J. Yu, Prompting large language models with answer heuristics for knowledge-based visual question answering, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 14974\u201314983.","DOI":"10.1109\/CVPR52729.2023.01438"},{"key":"10.1016\/j.patcog.2026.114161_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110941","article-title":"M3ixup: A multi-modal data augmentation approach for image captioning","volume":"158","author":"Li","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114161_b5","doi-asserted-by":"crossref","first-page":"3014","DOI":"10.1109\/TASLP.2024.3407571","article-title":"Exploring clean label backdoor attacks and defense in language models","volume":"32","author":"Zhao","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.patcog.2026.114161_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112105","article-title":"Artwork protection against unauthorized neural style transfer and aesthetic color distance metric","volume":"171","author":"Guo","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114161_b7","first-page":"23887","article-title":"Learning from mistakes: Self-correct adversarial training for Chinese unnatural text correction","volume":"vol. 39","author":"Feng","year":"2025"},{"key":"10.1016\/j.patcog.2026.114161_b8","first-page":"25543","article-title":"Detecting and mitigating hallucination in large vision language models via fine-grained ai feedback","volume":"vol. 39","author":"Xiao","year":"2025"},{"key":"10.1016\/j.patcog.2026.114161_b9","first-page":"1","article-title":"Affective-ROPTester: Capability and bias analysis of LLMs in predicting retinopathy of prematurity","author":"Zhao","year":"2025","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.patcog.2026.114161_b10","series-title":"Trustworthy LLMs: a survey and guideline for evaluating large language models\u2019 alignment","author":"Liu","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b11","series-title":"Safety assessment of chinese large language models","author":"Sun","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110959","article-title":"Multimodal self-supervised learning for remote sensing data land cover classification","volume":"157","author":"Xue","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114161_b13","first-page":"1","article-title":"UniFLE: Uniform fusion of multiple LoRA experts for backdoor defense in large language models","author":"Zhao","year":"2026","journal-title":"IEEE Trans. Dependable Secur. Comput."},{"key":"10.1016\/j.patcog.2026.114161_b14","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny, MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models, in: The Twelfth International Conference on Learning Representations, 2023, pp. 1\u201317."},{"key":"10.1016\/j.patcog.2026.114161_b15","first-page":"1","article-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","volume":"36","author":"Dai","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114161_b16","series-title":"Llama-adapter v2: Parameter-efficient visual instruction model","author":"Gao","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b17","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: a visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114161_b18","series-title":"Self-debias: Self-correcting for debiasing large language models","author":"Feng","year":"2026"},{"key":"10.1016\/j.patcog.2026.114161_b19","first-page":"1","article-title":"Multimodal chain-of-thought reasoning in language models","volume":"2024","author":"Zhang","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.114161_b20","series-title":"C2PO: Diagnosing and disentangling bias shortcuts in LLMs","author":"Feng","year":"2025"},{"key":"10.1016\/j.patcog.2026.114161_b21","doi-asserted-by":"crossref","unstructured":"H. Xia, Q. Dong, L. Li, J. Xu, T. Liu, Z. Qin, Z. Sui, ImageNetVC: Zero-and Few-Shot Visual Commonsense Evaluation on 1000 ImageNet Categories, in: The 2023 Conference on Empirical Methods in Natural Language Processing, 2023, pp. 2009\u20132026.","DOI":"10.18653\/v1\/2023.findings-emnlp.133"},{"issue":"3","key":"10.1016\/j.patcog.2026.114161_b22","doi-asserted-by":"crossref","first-page":"1877","DOI":"10.1109\/TPAMI.2024.3507000","article-title":"LVLM-EHub: A comprehensive evaluation benchmark for large vision-language models","volume":"47","author":"Xu","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114161_b23","unstructured":"C. Fu, P. Chen, Y. Shen, Y. Qin, M. Zhang, X. Lin, J. Yang, X. Zheng, K. Li, X. Sun, Y. Wu, R. Ji, C. Shan, R. He, MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models, in: The Thirty-Ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track, 2025, pp. 1\u201319."},{"key":"10.1016\/j.patcog.2026.114161_b24","doi-asserted-by":"crossref","unstructured":"Y. Liu, H. Duan, Y. Zhang, B. Li, S. Zhang, W. Zhao, Y. Yuan, J. Wang, C. He, Z. Liu, et al., Mmbench: Is your multi-modal model an all-around player?, in: European Conference on Computer Vision, 2024, pp. 216\u2013233.","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"10.1016\/j.patcog.2026.114161_b25","series-title":"European Conference on Computer Vision","first-page":"386","article-title":"Mm-safetybench: A benchmark for safety evaluation of multimodal large language models","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b26","series-title":"Assessment of multimodal large language models in alignment with human values","author":"Shi","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b27","unstructured":"L. Rice, E. Wong, Z. Kolter, Overfitting in adversarially robust deep learning, in: International Conference on Machine Learning, 2020, pp. 8093\u20138104."},{"key":"10.1016\/j.patcog.2026.114161_b28","unstructured":"H. Zhang, Y. Yu, J. Jiao, E. Xing, L. El Ghaoui, M. Jordan, Theoretically principled trade-off between robustness and accuracy, in: International Conference on Machine Learning, 2019, pp. 7472\u20137482."},{"key":"10.1016\/j.patcog.2026.114161_b29","first-page":"2958","article-title":"Adversarial weight perturbation helps robust generalization","volume":"33","author":"Wu","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"2","key":"10.1016\/j.patcog.2026.114161_b30","first-page":"897","article-title":"Two-step strategy for domain adaptation retrieval","volume":"36","author":"Chen","year":"2023","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.patcog.2026.114161_b31","series-title":"2024 IEEE 40th International Conference on Data Engineering","first-page":"3111","article-title":"Denoising high-order graph clustering","author":"Chen","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b32","series-title":"Qwen-VL: A versatile vision-language model for understanding, localization, text reading, and beyond","author":"Bai","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b33","unstructured":"J. Li, D. Li, S. Savarese, S. Hoi, Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models, in: International Conference on Machine Learning, 2023, pp. 19730\u201319742."},{"key":"10.1016\/j.patcog.2026.114161_b34","doi-asserted-by":"crossref","first-page":"121475","DOI":"10.52202\/079017-3860","article-title":"Cogvlm: Visual expert for pretrained language models","volume":"37","author":"Wang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114161_b35","doi-asserted-by":"crossref","unstructured":"Z. Chen, J. Wu, W. Wang, W. Su, G. Chen, S. Xing, M. Zhong, Q. Zhang, X. Zhu, L. Lu, et al., Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 24185\u201324198.","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"10.1016\/j.patcog.2026.114161_b36","series-title":"Visual chatgpt: Talking, drawing and editing with visual foundation models","author":"Wu","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b37","series-title":"mplug-owl: Modularization empowers large language models with multimodality","author":"Ye","year":"2023"},{"key":"10.1016\/j.patcog.2026.114161_b38","doi-asserted-by":"crossref","unstructured":"Q. Ye, H. Xu, J. Ye, M. Yan, A. Hu, H. Liu, Q. Qian, J. Zhang, F. Huang, mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13040\u201313051.","DOI":"10.1109\/CVPR52733.2024.01239"},{"key":"10.1016\/j.patcog.2026.114161_b39","series-title":"Internlm-xcomposer2: Mastering free-form text-image composition and comprehension in vision-language large model","author":"Dong","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b40","series-title":"Moe-llava: Mixture of experts for large vision-language models","author":"Lin","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b41","doi-asserted-by":"crossref","unstructured":"Z. Li, B. Yang, Q. Liu, Z. Ma, S. Zhang, J. Yang, Y. Sun, Y. Liu, X. Bai, Monkey: Image resolution and text label are important things for large multi-modal models, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26763\u201326773.","DOI":"10.1109\/CVPR52733.2024.02527"},{"key":"10.1016\/j.patcog.2026.114161_b42","doi-asserted-by":"crossref","unstructured":"J. Cha, W. Kang, J. Mun, B. Roh, Honeybee: Locality-enhanced projector for multimodal llm, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13817\u201313827.","DOI":"10.1109\/CVPR52733.2024.01311"},{"key":"10.1016\/j.patcog.2026.114161_b43","unstructured":"H. Zhao, Z. Cai, S. Si, X. Ma, K. An, L. Chen, Z. Liu, S. Wang, W. Han, B. Chang, MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning, in: The Twelfth International Conference on Learning Representations, 2024, pp. 1\u201339."},{"key":"10.1016\/j.patcog.2026.114161_b44","series-title":"Minicpm: Unveiling the potential of small language models with scalable training strategies","author":"Hu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114161_b45","unstructured":"J. Hu, Y. Yao, C. Wang, S. WANG, Y. Pan, Q. Chen, T. Yu, H. Wu, Y. Zhao, H. Zhang, X. Han, Y. Lin, J. Xue, dahai li, Z. Liu, M. Sun, Large Multilingual Models Pivot Zero-Shot Multimodal Learning across Languages, in: The Twelfth International Conference on Learning Representations, 2024, pp. 1\u201326."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601126X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032601126X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T20:25:27Z","timestamp":1783023927000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032601126X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":45,"alternative-id":["S003132032601126X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114161","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A hierarchical evaluation framework for security and trustworthiness in Large Vision-Language Models","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114161","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114161"}}