{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T10:12:28Z","timestamp":1783764748013,"version":"3.55.0"},"reference-count":45,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472332"],"award-info":[{"award-number":["62472332"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114380","type":"journal-article","created":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T15:08:26Z","timestamp":1783177706000},"page":"114380","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["Mitigating hallucination in Multimodal Large Language Models via cross-layer visual anchors"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0936-0681","authenticated-orcid":false,"given":"Chengxu","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siqi","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7924-8620","authenticated-orcid":false,"given":"Jingling","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuang","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114380_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2026.113369","article-title":"Multimodal metaphor understanding and generation in MLLMs: A dataset and reasoning framework","volume":"178","author":"Yang","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114380_b2","article-title":"Individual time series and composite forecasting of the Chinese stock index","volume":"5","author":"Xu","year":"2021","journal-title":"Mach. Learn. Appl."},{"key":"10.1016\/j.patcog.2026.114380_b3","doi-asserted-by":"crossref","DOI":"10.1007\/s44443-026-00605-w","article-title":"A lightweight model for indoor object detection in unstructured scenes based on joint attention and prior knowledge in the context of home rehabilitation","author":"Xing","year":"2026","journal-title":"J. King Saud Univ. Comput. Inf. Sci"},{"key":"10.1016\/j.patcog.2026.114380_b4","article-title":"Human-computer interactive rehabilitation: A 3D graph deep learning method for non-contact gesture recognition in post-epidemic and aging societies","author":"Xing","year":"2025","journal-title":"Measurement"},{"key":"10.1016\/j.patcog.2026.114380_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112538","article-title":"RMPT: Retrieval-based multimodal prompt tuning for event detection","volume":"172","author":"Zhao","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114380_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2026.113456","article-title":"Multimodal artificial intelligence for disease diagnosis: Advances, applications, and challenges","volume":"178","author":"Cai","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114380_b7","doi-asserted-by":"crossref","unstructured":"Q. Huang, X. Dong, P. Zhang, B. Wang, C. He, J. Wang, D. Lin, W. Zhang, N. Yu, Opera: Alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13418\u201313427.","DOI":"10.1109\/CVPR52733.2024.01274"},{"key":"10.1016\/j.patcog.2026.114380_b8","first-page":"6434","article-title":"Convis: Contrastive decoding with hallucination visualization for mitigating hallucinations in multimodal large language models","volume":"vol. 39","author":"Park","year":"2025"},{"key":"10.1016\/j.patcog.2026.114380_b9","doi-asserted-by":"crossref","unstructured":"Z. Wan, C. Zhang, S. Yong, M.Q. Ma, S. Stepputtis, L.-P. Morency, D. Ramanan, K. Sycara, Y. Xie, Only: One-layer intervention sufficiently mitigates hallucinations in large vision-language models, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 3225\u20133234.","DOI":"10.1109\/ICCV51701.2025.00309"},{"key":"10.1016\/j.patcog.2026.114380_b10","first-page":"10306","article-title":"ASCD: Attention-steerable contrastive decoding for reducing hallucination in MLLM","volume":"vol. 40","author":"Wang","year":"2026"},{"key":"10.1016\/j.patcog.2026.114380_b11","doi-asserted-by":"crossref","unstructured":"H. Yin, G. Si, Z. Wang, ClearSight: Visual Signal Enhancement for Object Hallucination Mitigation in Multimodal Large Language Models, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 14625\u201314634.","DOI":"10.1109\/CVPR52734.2025.01363"},{"key":"10.1016\/j.patcog.2026.114380_b12","series-title":"Mitigating hallucination of large vision-language models via dynamic logits calibration","author":"Chen","year":"2025"},{"key":"10.1016\/j.patcog.2026.114380_b13","series-title":"International Conference on Machine Learning","first-page":"80873","article-title":"Look twice before you answer: Memory-space visual retracing for hallucination mitigation in Multimodal Large Language Models","author":"Zou","year":"2025"},{"key":"10.1016\/j.patcog.2026.114380_b14","first-page":"13712","article-title":"Mllm can see? dynamic correction decoding for hallucination mitigation","volume":"vol. 2025","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.114380_b15","first-page":"35799","article-title":"The hidden life of tokens: Reducing hallucination of large vision-language models via visual information steering","volume":"267","author":"Li","year":"2025","journal-title":"Proc. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.114380_b16","unstructured":"J. Zhang, M. Khayatkhoei, P. Chhikara, F. Ilievski, MLLMs Know Where to Look: Training-free Perception of Small Visual Details with Multimodal LLMs, in: The Thirteenth International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.114380_b17","doi-asserted-by":"crossref","unstructured":"Y. Li, H. Wang, X. Ding, H. Wang, X. Li, Token activation map to visually explain multimodal llms, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 48\u201358.","DOI":"10.1109\/ICCV51701.2025.00012"},{"key":"10.1016\/j.patcog.2026.114380_b18","series-title":"International Conference on Machine Learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.patcog.2026.114380_b19","doi-asserted-by":"crossref","first-page":"49250","DOI":"10.52202\/075280-2142","article-title":"Instructblip: Towards general-purpose vision-language models with instruction tuning","volume":"36","author":"Dai","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114380_b20","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny, MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.patcog.2026.114380_b21","series-title":"The evolution of multimodal model architectures","author":"Wadekar","year":"2024"},{"key":"10.1016\/j.patcog.2026.114380_b22","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, Y. Li, Y.J. Lee, Improved baselines with visual instruction tuning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.patcog.2026.114380_b23","series-title":"Llavanext: Improved reasoning, ocr, and world knowledge","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114380_b24","doi-asserted-by":"crossref","first-page":"121475","DOI":"10.52202\/079017-3860","article-title":"Cogvlm: Visual expert for pretrained language models","volume":"37","author":"Wang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"2","key":"10.1016\/j.patcog.2026.114380_b25","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3703155","article-title":"A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions","volume":"43","author":"Huang","year":"2025","journal-title":"ACM Trans. Inf. Syst."},{"key":"10.1016\/j.patcog.2026.114380_b26","unstructured":"C. Yang, J. Yuan, S. Cai, J. Jiang, C. Hu, Heaven-Sent or Hell-Bent? Benchmarking the Intelligence and Defectiveness of LLM Hallucinations, in: Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1, 2026, pp. 2830\u20132841."},{"key":"10.1016\/j.patcog.2026.114380_b27","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Exploring prosocial irrationality for LLM agents: A social cognition view","author":"Liu","year":"2025"},{"key":"10.1016\/j.patcog.2026.114380_b28","doi-asserted-by":"crossref","unstructured":"Z. Jiang, J. Chen, B. Zhu, T. Luo, Y. Shen, X. Yang, Devils in middle layers of large vision-language models: Interpreting, detecting and mitigating object hallucinations via attention lens, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 25004\u201325014.","DOI":"10.1109\/CVPR52734.2025.02328"},{"key":"10.1016\/j.patcog.2026.114380_b29","unstructured":"Y. Cho, K. Kim, T. Hwang, S. Cho, Do You Keep an Eye on What I Ask? Mitigating Multimodal Hallucination via Attention-Guided Ensemble Decoding, in: The Thirteenth International Conference on Learning Representations."},{"key":"10.1016\/j.patcog.2026.114380_b30","doi-asserted-by":"crossref","unstructured":"L. Zhu, T. Chen, Q. Xu, X. Liu, D. Ji, H. Wu, D.W. Soh, J. Liu, Popen: Preference-based optimization and ensemble for lvlm-based reasoning segmentation, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 30231\u201330240.","DOI":"10.1109\/CVPR52734.2025.02814"},{"issue":"6","key":"10.1016\/j.patcog.2026.114380_b31","doi-asserted-by":"crossref","first-page":"4971","DOI":"10.1007\/s00521-024-10726-w","article-title":"Machine learning-based forecasts of residential property prices in hangzhou city, zhejiang province, China","volume":"37","author":"Jin","year":"2025","journal-title":"Neural Comput. Appl."},{"issue":"4","key":"10.1016\/j.patcog.2026.114380_b32","doi-asserted-by":"crossref","first-page":"1297","DOI":"10.1002\/ajae.12041","article-title":"Corn cash price forecasting","volume":"102","author":"Xu","year":"2020","journal-title":"Am. J. Agric. Econ."},{"key":"10.1016\/j.patcog.2026.114380_b33","doi-asserted-by":"crossref","unstructured":"S. Leng, H. Zhang, G. Chen, X. Li, S. Lu, C. Miao, L. Bing, Mitigating object hallucinations in large vision-language models through visual contrastive decoding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13872\u201313882.","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"10.1016\/j.patcog.2026.114380_b34","series-title":"Findings of the Association for Computational Linguistics ACL 2024","first-page":"15840","article-title":"Mitigating hallucinations in large vision-language models with instruction contrastive decoding","author":"Wang","year":"2024"},{"key":"10.1016\/j.patcog.2026.114380_b35","unstructured":"J. Li, J. Zhang, Z. Jie, L. Ma, M. Li, X. Luo, G. Li, Cross-Modal Attention Calibration for LVLM Hallucination Mitigation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2026, pp. 40186\u201340196."},{"key":"10.1016\/j.patcog.2026.114380_b36","first-page":"9439","article-title":"Not all tokens and heads are equally important: Dual-level attention intervention for hallucination mitigation","volume":"vol. 40","author":"Tang","year":"2026"},{"key":"10.1016\/j.patcog.2026.114380_b37","series-title":"R0-FoMo: Robustness of Few-Shot and Zero-Shot Learning in Large Foundation Models","article-title":"Visual cropping improves zero-shot question answering of multimodal large language models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.patcog.2026.114380_b38","doi-asserted-by":"crossref","unstructured":"F. Tang, C. Liu, Z. Xu, M. Hu, Z. Huang, H. Xue, Z. Chen, Z. Peng, Z. Yang, S. Zhou, et al., Seeing far and clearly: Mitigating hallucinations in mllms with attention causal decoding, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 26147\u201326159.","DOI":"10.1109\/CVPR52734.2025.02435"},{"key":"10.1016\/j.patcog.2026.114380_b39","series-title":"European Conference on Computer Vision","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.114380_b40","doi-asserted-by":"crossref","unstructured":"Z. Zhang, H. Tang, J. Sheng, Z. Zhang, Y. Ren, Z. Li, D. Yin, D. Ma, T. Liu, Debiasing multimodal large language models via noise-aware preference optimization, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 9423\u20139433.","DOI":"10.1109\/CVPR52734.2025.00880"},{"key":"10.1016\/j.patcog.2026.114380_b41","doi-asserted-by":"crossref","unstructured":"S. Leng, H. Zhang, G. Chen, X. Li, S. Lu, C. Miao, L. Bing, Mitigating object hallucinations in large vision-language models through visual contrastive decoding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13872\u201313882.","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"10.1016\/j.patcog.2026.114380_b42","series-title":"Qwen-vl: A frontier large vision-language model with versatile abilities","first-page":"3","author":"Bai","year":"2023"},{"key":"10.1016\/j.patcog.2026.114380_b43","doi-asserted-by":"crossref","unstructured":"Y. Li, Y. Du, K. Zhou, J. Wang, W.X. Zhao, J.-R. Wen, Evaluating Object Hallucination in Large Vision-Language Models, in: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 2023, pp. 292\u2013305.","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"10.1016\/j.patcog.2026.114380_b44","doi-asserted-by":"crossref","unstructured":"A. Rohrbach, L.A. Hendricks, K. Burns, T. Darrell, K. Saenko, Object Hallucination in Image Captioning, in: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, 2018, pp. 4035\u20134045.","DOI":"10.18653\/v1\/D18-1437"},{"key":"10.1016\/j.patcog.2026.114380_b45","unstructured":"C. Fu, P. Chen, Y. Shen, Y. Qin, M. Zhang, X. Lin, J. Yang, X. Zheng, K. Li, X. Sun, et al., Mme: A comprehensive evaluation benchmark for multimodal large language models, in: The Thirty-Ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track, 2025."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326013452?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326013452?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T09:41:07Z","timestamp":1783762867000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326013452"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":45,"alternative-id":["S0031320326013452"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114380","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Mitigating hallucination in Multimodal Large Language Models via cross-layer visual anchors","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114380","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114380"}}