{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T16:51:59Z","timestamp":1781542319175,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810847","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2022-2031","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Mitigating Hallucinations in Vision-Language Models via Contextual Entropy Calibration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0049-6328","authenticated-orcid":false,"given":"Xuanyu","family":"Yin","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5208-3811","authenticated-orcid":false,"given":"Daowan","family":"Peng","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4488-0102","authenticated-orcid":false,"given":"Wei","family":"Wei","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Jinze Bai Shuai Bai Yunfei Chu Zeyu Cui Kai Dang Xiaodong Deng Yang Fan Wenbin Ge Yu Han Fei Huang et\u00a0al. 2023. Qwen technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.16609 (2023)."},{"key":"e_1_3_3_1_3_2","unstructured":"Jinze Bai Shuai Bai Shusheng Yang Shijie Wang Sinan Tan Peng Wang Junyang Lin Chang Zhou and Jingren Zhou. 2023. Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.12966 (2023)."},{"key":"e_1_3_3_1_4_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang Humen Zhong Yuanzhi Zhu Mingkun Yang Zhaohai Li Jianqiang Wan Pengfei Wang Wei Ding Zheren Fu Yiheng Xu Jiabo Ye Xi Zhang Tianbao Xie Zesen Cheng Hang Zhang Zhibo Yang Haiyang Xu and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_1_5_2","unstructured":"Zechen Bai Pichao Wang Tianjun Xiao Tong He Zongbo Han Zheng Zhang and Mike\u00a0Zheng Shou. 2024. Hallucination of multimodal large language models: A survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.18930 (2024)."},{"key":"e_1_3_3_1_6_2","first-page":"335","volume-title":"ISMIR","author":"Boulanger-Lewandowski Nicolas","year":"2013","unstructured":"Nicolas Boulanger-Lewandowski, Yoshua Bengio, and Pascal Vincent. 2013. Audio Chord Recognition with Recurrent Neural Networks. In ISMIR. Curitiba, 335\u2013340."},{"key":"e_1_3_3_1_7_2","unstructured":"Keqin Chen Zhao Zhang Weili Zeng Richong Zhang Feng Zhu and Rui Zhao. 2023. Shikra: Unleashing multimodal llm\u2019s referential dialogue magic. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.15195 (2023)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611018"},{"key":"e_1_3_3_1_9_2","series-title":"(ICML\u201924)","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Chen Zhaorun","year":"2024","unstructured":"Zhaorun Chen, Zhuokai Zhao, Hongyin Luo, Huaxiu Yao, Bo Li, and Jiawei Zhou. 2024. HALC: object hallucination reduction via adaptive focal-contrast decoding. In Proceedings of the 41st International Conference on Machine Learning (Vienna, Austria) (ICML\u201924). JMLR.org, Article 307, 23\u00a0pages."},{"key":"e_1_3_3_1_10_2","unstructured":"Wei-Lin Chiang Zhuohan Li Ziqing Lin Ying Sheng Zhanghao Wu Hao Zhang Lianmin Zheng Siyuan Zhuang Yonghao Zhuang Joseph\u00a0E Gonzalez et\u00a0al. 2023. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys. org (accessed 14 April 2023) 2 3 (2023) 6."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.84"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Wenliang Dai Junnan Li Dongxu Li Anthony Tiong Junqi Zhao Weisheng Wang Boyang Li Pascale\u00a0N Fung and Steven Hoi. 2023. Instructblip: Towards general-purpose vision-language models with instruction tuning. Advances in neural information processing systems 36 (2023) 49250\u201349267.","DOI":"10.52202\/075280-2142"},{"key":"e_1_3_3_1_13_2","unstructured":"Xiaoyi Dong Pan Zhang Yuhang Zang Yuhang Cao Bin Wang Linke Ouyang Xilin Wei Songyang Zhang Haodong Duan Maosong Cao et\u00a0al. 2024. Internlm-xcomposer2: Mastering free-form text-image composition and comprehension in vision-language large model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.16420 (2024)."},{"key":"e_1_3_3_1_14_2","unstructured":"Chaoyou Fu Peixian Chen Yunhang Shen Yulei Qin Mengdan Zhang Xu Lin Jinrui Yang Xiawu Zheng Ke Li Xing Sun et\u00a0al. 2023. Mme: A comprehensive evaluation benchmark for multimodal large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.13394 (2023)."},{"key":"e_1_3_3_1_15_2","unstructured":"Alex Graves. 2012. Sequence transduction with recurrent neural networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1211.3711 (2012)."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731715.3733374"},{"key":"e_1_3_3_1_17_2","unstructured":"Mingzhe Hu Shaoyan Pan Yuheng Li and Xiaofeng Yang. 2023. Advancing medical imaging with language models: A journey from n-grams to chatgpt. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.04920 (2023)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Qidong Huang Xiaoyi Dong Pan Zhang Bin Wang Conghui He Jiaqi Wang Dahua Lin Weiming Zhang and Nenghai Yu. 2024. Opera: Alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2024).","DOI":"10.1109\/CVPR52733.2024.01274"},{"key":"e_1_3_3_1_19_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Huo Fushuo","year":"2025","unstructured":"Fushuo Huo, Wenchao Xu, Zhong Zhang, Haozhao Wang, Zhicheng Chen, and Peilin Zhao. 2025. Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=rsZwwjYHuD"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Ziwei Ji Nayeon Lee Rita Frieske Tiezheng Yu Dan Su Yan Xu Etsuko Ishii Ye\u00a0Jin Bang Andrea Madotto and Pascale Fung. 2023. Survey of hallucination in natural language generation. ACM computing surveys 55 12 (2023) 1\u201338.","DOI":"10.1145\/3571730"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Xun Jiang Xing Xu Zailei Zhou Yang Yang Fumin Shen and Heng\u00a0Tao Shen. 2024. Zero-Shot Video Moment Retrieval With Angular Reconstructive Text Embeddings. IEEE Transactions on Multimedia 26 (2024) 9657\u20139670.","DOI":"10.1109\/TMM.2024.3396272"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731715.3734581"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.23"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Yunxin Li Baotian Hu Xinyu Chen Lin Ma Yong Xu and Min Zhang. 2024. LMEye: An Interactive Perception Network for Large Language Models. IEEE Transactions on Multimedia 26 (2024) 10952\u201310964.","DOI":"10.1109\/TMM.2024.3428317"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_3_1_28_2","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024","author":"Liu Fuxiao","year":"2024","unstructured":"Fuxiao Liu, Kevin Lin, Linjie Li, Jianfeng Wang, Yaser Yacoob, and Lijuan Wang. 2024. Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=J44HfH4JCg"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual instruction tuning. Advances in neural information processing systems 36 (2023) 34892\u201334916.","DOI":"10.52202\/075280-1516"},{"key":"e_1_3_3_1_31_2","unstructured":"Hanchao Liu Wenyuan Xue Yifei Chen Dapeng Chen Xiutian Zhao Ke Wang Liping Hou Rongjun Li and Wei Peng. 2024. A survey on hallucination in large vision-language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.00253 (2024)."},{"key":"e_1_3_3_1_32_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Liu Sheng","year":"2025","unstructured":"Sheng Liu, Haotian Ye, and James Zou. 2025. Reducing Hallucinations in Large Vision-Language Models via Latent Space Steering. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025."},{"key":"e_1_3_3_1_33_2","first-page":"125","volume-title":"European Conference on Computer Vision","author":"Liu Shi","year":"2024","unstructured":"Shi Liu, Kecheng Zheng, and Wei Chen. 2024. Paying more attention to image: A training-free method for alleviating hallucination in lvlms. In European Conference on Computer Vision. Springer, 125\u2013140."},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1437"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"e_1_3_3_1_36_2","unstructured":"Ilya Sutskever Oriol Vinyals and Quoc\u00a0V Le. 2014. Sequence to sequence learning with neural networks. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_3_1_37_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar et\u00a0al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.13971 (2023)."},{"key":"e_1_3_3_1_38_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et\u00a0al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_1_39_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Wang Chenxi","year":"2025","unstructured":"Chenxi Wang, Xiang Chen, Ningyu Zhang, Bozhong Tian, Haoming Xu, Shumin Deng, and Huajun Chen. 2025. MLLM can see? Dynamic Correction Decoding for Hallucination Mitigation. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=4z3IguA4Zg"},{"key":"e_1_3_3_1_40_2","unstructured":"Junyang Wang Yuhang Wang Guohai Xu Jing Zhang Yukai Gu Haitao Jia Ming Yan Ji Zhang and Jitao Sang. 2023. An LLM-free Multi-dimensional Benchmark for MLLMs Hallucination Evaluation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.07397 (2023)."},{"key":"e_1_3_3_1_41_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Wang Kaishen","year":"2025","unstructured":"Kaishen Wang, Hengrui Gu, Meijun Gao, and Kaixiong Zhou. 2025. DAMO: Decoding by Accumulating Activations Momentum for Mitigating Hallucinations in Vision-Language Models. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"crossref","unstructured":"Sheng Wang Zihao Zhao Xi Ouyang Tianming Liu Qian Wang and Dinggang Shen. 2024. Interactive computer-aided diagnosis on medical image using large language models. Communications Engineering 3 1 (2024) 133.","DOI":"10.1038\/s44172-024-00271-8"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.937"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"crossref","unstructured":"Ziqiang Wu Bingpeng Ma Hong Chang and Shiguang Shan. 2023. Refined Knowledge Transfer for Language-Based Person Search. IEEE Transactions on Multimedia 25 (2023) 9315\u20139329.","DOI":"10.1109\/TMM.2023.3251104"},{"key":"e_1_3_3_1_45_2","unstructured":"Zhenyu Wu Ziwei Wang Xiuwei Xu Jiwen Lu and Haibin Yan. 2023. Embodied Task Planning with Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.01848 (2023)."},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3731715.3733450"},{"key":"e_1_3_3_1_47_2","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei Huan Lin Jian Yang Jianhong Tu Jianwei Zhang Jianxin Yang Jiaxi Yang Jingren Zhou Junyang Lin Kai Dang Keming Lu Keqin Bao Kexin Yang Le Yu Mei Li Mingfeng Xue Pei Zhang Qin Zhu Rui Men Runji Lin Tianhao Li Tingyu Xia Xingzhang Ren Xuancheng Ren Yang Fan Yang Su Yichang Zhang Yu Wan Yuqiong Liu Zeyu Cui Zhenru Zhang and Zihan Qiu. 2024. Qwen2.5 Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.15115 (2024)."},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"crossref","unstructured":"Xingyu Yang Mengya Han Yong Luo Han Hu and Yonggang Wen. 2023. Two-Stream Prototype Learning Network for Few-Shot Face Recognition Under Occlusions. IEEE Transactions on Multimedia 25 (2023) 1555\u20131563.","DOI":"10.1109\/TMM.2023.3253054"},{"key":"e_1_3_3_1_49_2","unstructured":"Qinghao Ye Haiyang Xu Guohai Xu Jiabo Ye Ming Yan Yiyang Zhou Junyang Wang Anwen Hu Pengcheng Shi Yaya Shi Chenliang Li Yuanhong Xu Hehong Chen Junfeng Tian Qian Qi Ji Zhang and Fei Huang. 2023. mPLUG-Owl: Modularization Empowers Large Language Models with Multimodality. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.14178 (2023)."},{"key":"e_1_3_3_1_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01363"},{"key":"e_1_3_3_1_51_2","doi-asserted-by":"publisher","unstructured":"Shukang Yin Chaoyou Fu Sirui Zhao Tong Xu Hao Wang Dianbo Sui Yunhang Shen Ke Li Xing Sun and Enhong Chen. 2024. Woodpecker: hallucination correction for multimodal large language models. Sci. China Inf. Sci. 67 12 (2024). 10.1007\/S11432-024-4251-X","DOI":"10.1007\/S11432-024-4251-X"},{"key":"e_1_3_3_1_52_2","unstructured":"Qifan Yu Juncheng Li Longhui Wei Liang Pang Wentao Ye Bosheng Qin Siliang Tang Qi Tian and Yueting Zhuang. 2024. Hallucidoctor: Mitigating hallucinatory toxicity in visual instruction data. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2024)."},{"key":"e_1_3_3_1_53_2","unstructured":"Pan Zhang Xiaoyi Dong\u00a0Bin Wang Yuhang Cao Chao Xu Linke Ouyang Zhiyuan Zhao Shuangrui Ding Songyang Zhang Haodong Duan Hang Yan et\u00a0al. 2023. Internlm-xcomposer: A vision-language large model for advanced text-image comprehension and composition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.15112 (2023)."},{"key":"e_1_3_3_1_54_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.58"},{"key":"e_1_3_3_1_55_2","doi-asserted-by":"publisher","unstructured":"Xiaofeng Zhang Yihao Quan Chaochen Gu Chen Shen Xiaosong Yuan Shaotian Yan Hao Cheng Kaijie Wu and Jieping Ye. 2024. Seeing Clearly by Layer Two: Enhancing Attention Heads to Alleviate Hallucination in LVLMs. CoRR abs\/2411.09968 (2024). arXiv:https:\/\/arXiv.org\/abs\/2411.0996810.48550\/ARXIV.2411.09968","DOI":"10.48550\/ARXIV.2411.09968"},{"key":"e_1_3_3_1_56_2","unstructured":"Yiyang Zhou Chenhang Cui Rafael Rafailov Chelsea Finn and Huaxiu Yao. 2024. Aligning modalities in vision large language models via preference fine-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.11411 (2024)."},{"key":"e_1_3_3_1_57_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Zhou Yiyang","year":"2024","unstructured":"Yiyang Zhou, Chenhang Cui, Jaehong Yoon, Linjun Zhang, Zhun Deng, Chelsea Finn, Mohit Bansal, and Huaxiu Yao. 2024. Analyzing and Mitigating Object Hallucination in Large Vision-Language Models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_1_58_2","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=1tZbq88f27"},{"key":"e_1_3_3_1_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00396"}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:52:44Z","timestamp":1781538764000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810847"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":58,"alternative-id":["10.1145\/3805622.3810847","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810847","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}