{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T21:40:54Z","timestamp":1783806054532,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":86,"publisher":"ACM","funder":[{"name":"The Major Science and Technology Plan Project on the Future Industry Fields of Xiamen City","award":["3502Z20241027"],"award-info":[{"award-number":["3502Z20241027"]}]},{"name":"The Unveiling and Leading Projects of Xiamen","award":["3502Z20241011"],"award-info":[{"award-number":["3502Z20241011"]}]},{"name":"The Open Project of the State Key Laboratory of Multimodal Artificial Intelligence Systems","award":["MAIS2024101"],"award-info":[{"award-number":["MAIS2024101"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755271","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:38Z","timestamp":1761377198000},"page":"3981-3990","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["MCA-LLaVA: Manhattan Causal Attention for Reducing Hallucination in Large Vision-Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-5054-9742","authenticated-orcid":false,"given":"Qiyan","family":"Zhao","sequence":"first","affiliation":[{"name":"FKLPRIU, Xiamen University of Technology, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7185-4682","authenticated-orcid":false,"given":"Xiaofeng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9154-0458","authenticated-orcid":false,"given":"Yiheng","family":"Li","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9839-0120","authenticated-orcid":false,"given":"Yun","family":"Xing","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5748-5174","authenticated-orcid":false,"given":"Xiaosong","family":"Yuan","sequence":"additional","affiliation":[{"name":"Jilin University, Changchun, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7583-482X","authenticated-orcid":false,"given":"Feilong","family":"Tang","sequence":"additional","affiliation":[{"name":"Monash University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5186-8834","authenticated-orcid":false,"given":"Sinan","family":"Fan","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6000-3914","authenticated-orcid":false,"given":"Xuhang","family":"Chen","sequence":"additional","affiliation":[{"name":"Huizhou University, Huizhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5901-0778","authenticated-orcid":false,"given":"Da-Han","family":"Wang","sequence":"additional","affiliation":[{"name":"FKLPRIU, Xiamen University of Technology, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9260-188X","authenticated-orcid":false,"given":"Xu-Yao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention. arXiv preprint arXiv:2406.12718","author":"An Wenbin","year":"2024","unstructured":"Wenbin An, Feng Tian, Sicong Leng, Jiahao Nie, Haonan Lin, QianYing Wang, Guang Dai, Ping Chen, and Shijian Lu. 2024. AGLA: Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention. arXiv preprint arXiv:2406.12718 (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"Qwen-vl: A frontier large visionlanguage model with versatile abilities. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, ShijieWang, Sinan Tan, PengWang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-vl: A frontier large visionlanguage model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"Hallucination of Multimodal Large Language Models: A Survey. ArXiv abs\/2404.18930","author":"Bai Zechen","year":"2024","unstructured":"Zechen Bai, Pichao Wang, Tianjun Xiao, Tong He, Zongbo Han, Zheng Zhang, and Mike Zheng Shou. 2024. Hallucination of Multimodal Large Language Models: A Survey. ArXiv abs\/2404.18930 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"Alleviating Hallucinations in Large Vision-Language Models through Hallucination- Induced Optimization. ArXiv abs\/2405.15356","author":"Chen Beitao","year":"2024","unstructured":"Beitao Chen, Xinyu Lyu, Lianli Gao, Jingkuan Song, and Hengtao Shen. 2024. Alleviating Hallucinations in Large Vision-Language Models through Hallucination- Induced Optimization. ArXiv abs\/2405.15356 (2024)."},{"key":"e_1_3_2_1_5_1","unstructured":"Cong Chen Mingyu Liu Chenchen Jing Yizhou Zhou Fengyun Rao Hao Chen Bo Zhang and Chunhua Shen. 2025. PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training."},{"key":"e_1_3_2_1_6_1","volume-title":"Pan Zhang, Yuhang Zang, Zehui Chen, Haodong Duan, Jiaqi Wang, Yu Qiao, Dahua Lin, and Feng Zhao.","author":"Chen Lin","year":"2024","unstructured":"Lin Chen, Jinsong Li, Xiao wen Dong, Pan Zhang, Yuhang Zang, Zehui Chen, Haodong Duan, Jiaqi Wang, Yu Qiao, Dahua Lin, and Feng Zhao. 2024. Are We on the Right Way for Evaluating Large Vision-Language Models? ArXiv abs\/2403.20330 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73004-7_2"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al. 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_1_9_1","volume-title":"HALC: Object Hallucination Reduction via Adaptive Focal-Contrast Decoding. arXiv preprint arXiv:2403.00425","author":"Chen Zhaorun","year":"2024","unstructured":"Zhaorun Chen, Zhuokai Zhao, Hongyin Luo, Huaxiu Yao, Bo Li, and Jiawei Zhou. 2024. HALC: Object Hallucination Reduction via Adaptive Focal-Contrast Decoding. arXiv preprint arXiv:2403.00425 (2024)."},{"key":"e_1_3_2_1_10_1","volume-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys.org (accessed","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E Gonzalez, et al. 2023. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys.org (accessed 14 April 2023) 2, 3 (2023), 6."},{"key":"e_1_3_2_1_11_1","volume-title":"Dola: Decoding by contrasting layers improves factuality in large language models. arXiv preprint arXiv:2309.03883","author":"Chuang Yung-Sung","year":"2023","unstructured":"Yung-Sung Chuang, Yujia Xie, Hongyin Luo, Yoon Kim, James Glass, and Pengcheng He. 2023. Dola: Decoding by contrasting layers improves factuality in large language models. arXiv preprint arXiv:2309.03883 (2023)."},{"key":"e_1_3_2_1_12_1","volume-title":"Fine-Grained Verifiers: Preference Modeling as Next-token Prediction in Vision-Language Alignment. ArXiv abs\/2410.14148","author":"Cui Chenhang","year":"2024","unstructured":"Chenhang Cui, An Zhang, Yiyang Zhou, Zhaorun Chen, Gelei Deng, Huaxiu Yao, and Tat-Seng Chua. 2024. Fine-Grained Verifiers: Preference Modeling as Next-token Prediction in Vision-Language Alignment. ArXiv abs\/2410.14148 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_14_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ICLR","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ICLR (2021)."},{"key":"e_1_3_2_1_15_1","volume-title":"Mitigating Hallucination in Multimodal Large Language Model via Hallucinationtargeted Direct Preference Optimization. ArXiv abs\/2411.10436","author":"Fu Yuhan","year":"2024","unstructured":"Yuhan Fu, Ruobing Xie, Xingwu Sun, Zhanhui Kang, and Xirong Li. 2024. Mitigating Hallucination in Multimodal Large Language Model via Hallucinationtargeted Direct Preference Optimization. ArXiv abs\/2411.10436 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.439"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1116-0"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29771"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00380"},{"key":"e_1_3_2_1_20_1","volume-title":"Incorporating Visual Experts to Resolve the Information Loss in Multimodal Large Language Models. ArXiv abs\/2401.03105","author":"He Xin","year":"2024","unstructured":"Xin He, LonghuiWei, Lingxi Xie, and Qi Tian. 2024. Incorporating Visual Experts to Resolve the Information Loss in Multimodal Large Language Models. ArXiv abs\/2401.03105 (2024)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01274"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.298"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00686"},{"key":"e_1_3_2_1_24_1","volume-title":"Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models. In The Thirteenth International Conference on Learning Representations.","author":"Huo Fushuo","year":"2025","unstructured":"Fushuo Huo, Wenchao Xu, Zhong Zhang, Haozhao Wang, Zhicheng Chen, and Peilin Zhao. 2025. Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_25_1","first-page":"27026","volume-title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model. 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Jiang Chaoya","year":"2023","unstructured":"Chaoya Jiang, Haiyang Xu, Mengfan Dong, Jiaxing Chen, Wei Ye, Mingshi Yan, Qinghao Ye, Ji Zhang, Fei Huang, and Shikun Zhang. 2023. Hallucination Augmented Contrastive Learning for Multimodal Large Language Model. 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2023), 27026-27036."},{"key":"e_1_3_2_1_26_1","volume-title":"CODE: Contrasting Self-generated Description to Combat Hallucination in Large Multi-modal Models. ArXiv abs\/2406.01920","author":"Kim Junho","year":"2024","unstructured":"Junho Kim, Hyunjun Kim, Yeonju Kim, and Yonghyun Ro. 2024. CODE: Contrasting Self-generated Description to Combat Hallucination in Large Multi-modal Models. ArXiv abs\/2406.01920 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"VACoDe: Visual Augmented Contrastive Decoding. ArXiv abs\/2408.05337","author":"Kim Sihyeon","year":"2024","unstructured":"Sihyeon Kim, Boryeong Cho, Sangmin Bae, Sumyeong Ahn, and SeYoung Yun. 2024. VACoDe: Visual Augmented Contrastive Decoding. ArXiv abs\/2408.05337 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"The Curse of Multi-Modalities: Evaluating Hallucinations of Large Multimodal Models across Language, Visual, and Audio. ArXiv abs\/2410.12787","author":"Leng Sicong","year":"2024","unstructured":"Sicong Leng, Yun Xing, Zesen Cheng, Yang Zhou, Hang Zhang, Xin Li, Deli Zhao, Shijian Lu, Chunyan Miao, and Li Bing. 2024. The Curse of Multi-Modalities: Evaluating Hallucinations of Large Multimodal Models across Language, Visual, and Audio. ArXiv abs\/2410.12787 (2024)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"e_1_3_2_1_30_1","volume-title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension. ArXiv abs\/2307.16125","author":"Li Bohao","year":"2023","unstructured":"Bohao Li, RuiWang, GuangzhiWang, Yuying Ge, Yixiao Ge, and Ying Shan. 2023. SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension. ArXiv abs\/2307.16125 (2023)."},{"key":"e_1_3_2_1_31_1","volume-title":"Llava-med: Training a large language-and-vision assistant for biomedicine in one day. Advances in Neural Information Processing Systems 36","author":"Li Chunyuan","year":"2024","unstructured":"Chunyuan Li, Cliff Wong, Sheng Zhang, Naoto Usuyama, Haotian Liu, Jianwei Yang, Tristan Naumann, Hoifung Poon, and Jianfeng Gao. 2024. Llava-med: Training a large language-and-vision assistant for biomedicine in one day. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Inference-time intervention: Eliciting truthful answers from a language model. Advances in Neural Information Processing Systems 36","author":"Li Kenneth","year":"2024","unstructured":"Kenneth Li, Oam Patel, Fernanda Vi\u00e9gas, Hanspeter Pfister, and Martin Wattenberg. 2024. Inference-time intervention: Eliciting truthful answers from a language model. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_33_1","volume-title":"Wayne Xin Zhao, and Ji-Rong Wen","author":"Li Yifan","year":"2023","unstructured":"Yifan Li, Yifan Du, Kun Zhou, Jinpeng Wang, Wayne Xin Zhao, and Ji-Rong Wen. 2023. Evaluating object hallucination in large vision-language models. arXiv preprint arXiv:2305.10355 (2023)."},{"key":"e_1_3_2_1_34_1","volume-title":"Wayne Xin Zhao, and Ji-Rong Wen","author":"Li Yifan","year":"2023","unstructured":"Yifan Li, Yifan Du, Kun Zhou, Jinpeng Wang, Wayne Xin Zhao, and Ji-Rong Wen. 2023. Evaluating object hallucination in large vision-language models. arXiv preprint arXiv:2305.10355 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"Metaxas","author":"Li Zhuowei","year":"2025","unstructured":"Zhuowei Li, Haizhou Shi, Yunhe Gao, Di Liu, ZhentingWang, Yuxiao Chen, Ting Liu, Long Zhao, Hao Wang, and Dimitris N. Metaxas. 2025. The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering. ArXiv abs\/2502.03628 (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"International Conference on Learning Representations.","author":"Liu Fuxiao","year":"2023","unstructured":"Fuxiao Liu, Kevin Lin, Linjie Li, Jianfeng Wang, Yaser Yacoob, and Lijuan Wang. 2023. Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_37_1","first-page":"26286","volume-title":"Improved Baselines with Visual Instruction Tuning. 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Yuheng Li, and Yong Jae Lee. 2023. Improved Baselines with Visual Instruction Tuning. 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2023), 26286-26296."},{"key":"e_1_3_2_1_38_1","volume-title":"Visual Instruction Tuning. Advances in neural information processing systems 36","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, QingyangWu, and Yong Jae Lee. 2024. Visual Instruction Tuning. Advances in neural information processing systems 36 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"A Survey on Hallucination in Large Vision-Language Models. ArXiv abs\/2402.00253","author":"Liu Hanchao","year":"2024","unstructured":"Hanchao Liu, Wenyuan Xue, Yifei Chen, Dapeng Chen, Xiutian Zhao, Ke Wang, Liping Hou, Rong-Zhi Li, and Wei Peng. 2024. A Survey on Hallucination in Large Vision-Language Models. ArXiv abs\/2402.00253 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs. ArXiv abs\/2407.21771","author":"Liu Shiping","year":"2024","unstructured":"Shiping Liu, Kecheng Zheng, and Wei Chen. 2024. Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs. ArXiv abs\/2407.21771 (2024)."},{"key":"e_1_3_2_1_41_1","first-page":"2507","article-title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","volume":"35","author":"Lu Pan","year":"2022","unstructured":"Pan Lu, Swaroop Mishra, Tanglin Xia, Liang Qiu, Kai-Wei Chang, Song-Chun Zhu, Oyvind Tafjord, Peter Clark, and Ashwin Kalyan. 2022. Learn to explain: Multimodal reasoning via thought chains for science question answering. Advances in Neural Information Processing Systems 35 (2022), 2507-2521.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_42_1","first-page":"2199","volume-title":"DocVQA: A Dataset for VQA on Document Images. 2021 IEEE Winter Conference on Applications of Computer Vision (WACV)","author":"Mathew Minesh","year":"2020","unstructured":"Minesh Mathew, Dimosthenis Karatzas, R. Manmatha, and C. V. Jawahar. 2020. DocVQA: A Dataset for VQA on Document Images. 2021 IEEE Winter Conference on Applications of Computer Vision (WACV) (2020), 2199-2208."},{"key":"e_1_3_2_1_43_1","volume-title":"Strengthening Multimodal Large Language Model with Bootstrapped Preference Optimization. ArXiv abs\/2403.08730","author":"Pi Renjie","year":"2024","unstructured":"Renjie Pi, Tianyang Han, Wei Xiong, Jipeng Zhang, Runtao Liu, Rui Pan, and Tong Zhang. 2024. Strengthening Multimodal Large Language Model with Bootstrapped Preference Optimization. ArXiv abs\/2403.08730 (2024)."},{"key":"e_1_3_2_1_44_1","volume-title":"International conference on machine learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_45_1","volume-title":"Kaylee Burns, Trevor Darrell, and Kate Saenko.","author":"Rohrbach Anna","year":"2018","unstructured":"Anna Rohrbach, Lisa Anne Hendricks, Kaylee Burns, Trevor Darrell, and Kate Saenko. 2018. Object hallucination in image captioning. arXiv preprint arXiv:1809.02156 (2018)."},{"key":"e_1_3_2_1_46_1","unstructured":"Pritam Sarkar Sayna Ebrahimi Ali Etemad Ahmad Beirami Sercan \u00d6. Arik and Tomas Pfister. 2024. Mitigating Object Hallucination in MLLMs via Dataaugmented Phrase-level Alignment."},{"key":"e_1_3_2_1_47_1","volume-title":"From Pixels to Tokens: Revisiting Object Hallucinations in Large Vision-Language Models. ArXiv abs\/2410.06795","author":"Shang Yuying","year":"2024","unstructured":"Yuying Shang, Xinyi Zeng, Yutao Zhu, Xiao Yang, Zhengwei Fang, Jingyuan Zhang, Jiawei Chen, Zinan Liu, and Yu Tian. 2024. From Pixels to Tokens: Revisiting Object Hallucinations in Large Vision-Language Models. ArXiv abs\/2410.06795 (2024)."},{"key":"e_1_3_2_1_48_1","first-page":"8309","volume-title":"Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach. 2019. Towards VQA Models That Can Read. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Singh Amanpreet","year":"2019","unstructured":"Amanpreet Singh, Vivek Natarajan, Meet Shah, Yu Jiang, Xinlei Chen, Dhruv Batra, Devi Parikh, and Marcus Rohrbach. 2019. Towards VQA Models That Can Read. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019), 8309-8318."},{"key":"e_1_3_2_1_49_1","volume-title":"RoFormer: Enhanced Transformer with Rotary Position Embedding. ArXiv abs\/2104.09864","author":"Su Jianlin","year":"2021","unstructured":"Jianlin Su, Yu Lu, Shengfeng Pan, Bo Wen, and Yunfeng Liu. 2021. RoFormer: Enhanced Transformer with Rotary Position Embedding. ArXiv abs\/2104.09864 (2021)."},{"key":"e_1_3_2_1_50_1","unstructured":"Zhiqing Sun Sheng Shen Shengcao Cao Haotian Liu Chunyuan Li Yikang Shen Chuang Gan Liang-Yan Gui Yu-Xiong Wang Yiming Yang et al. 2023. Aligning large multimodal models with factually augmented rlhf. arXiv preprint arXiv:2309.14525 (2023)."},{"key":"e_1_3_2_1_51_1","volume-title":"The Thirteenth International Conference on Learning Representations.","author":"Tang Feilong","year":"2025","unstructured":"Feilong Tang, Zile Huang, Chengzhi Liu, Qiang Sun, Harry Yang, and Ser-Nam Lim. 2025. Intervening anchor token: Decoding strategy in alleviating hallucinations for MLLMs. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_52_1","volume-title":"Intervening Anchor Token: Decoding Strategy in Alleviating Hallucinations for MLLMs. In The Thirteenth International Conference on Learning Representations.","author":"Tang Feilong","year":"2025","unstructured":"Feilong Tang, Zile Huang, Chengzhi Liu, Qiang Sun, Harry Yang, and Ser-Nam Lim. 2025. Intervening Anchor Token: Decoding Strategy in Alleviating Hallucinations for MLLMs. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02435"},{"key":"e_1_3_2_1_54_1","volume-title":"LLaMA: Open and Efficient Foundation Language Models. ArXiv abs\/2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, Aur\u00e9lien Rodriguez, Armand Joulin, Edouard Grave, and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. ArXiv abs\/2302.13971 (2023)."},{"key":"e_1_3_2_1_55_1","unstructured":"Ashish Vaswani Noam M. Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In Neural Information Processing Systems."},{"key":"e_1_3_2_1_56_1","volume-title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models. ArXiv abs\/2406.11839","author":"Wang Fei","year":"2024","unstructured":"Fei Wang, Wenxuan Zhou, James Y. Huang, Nan Xu, Sheng Zhang, Hoifung Poon, and Muhao Chen. 2024. mDPO: Conditional Preference Optimization for Multimodal Large Language Models. ArXiv abs\/2406.11839 (2024)."},{"key":"e_1_3_2_1_57_1","volume-title":"VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding. ArXiv abs\/2411.15839","author":"Wang Jiaqi","year":"2024","unstructured":"Jiaqi Wang, Yifei Gao, and Jitao Sang. 2024. VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding. ArXiv abs\/2411.15839 (2024)."},{"key":"e_1_3_2_1_58_1","volume-title":"Conference on Multimedia Modeling.","author":"Wang Lei","year":"2023","unstructured":"Lei Wang, Jiabang He, Shenshen Li, Ning Liu, and Ee-Peng Lim. 2023. Mitigating Fine-Grained Hallucination by Fine-Tuning Large Vision-Language Models with Caption Rewrites. In Conference on Multimedia Modeling."},{"key":"e_1_3_2_1_59_1","volume-title":"Label words are anchors: An information flow perspective for understanding in-context learning. arXiv preprint arXiv:2305.14160","author":"Wang Lean","year":"2023","unstructured":"Lean Wang, Lei Li, Damai Dai, Deli Chen, Hao Zhou, Fandong Meng, Jie Zhou, and Xu Sun. 2023. Label words are anchors: An information flow perspective for understanding in-context learning. arXiv preprint arXiv:2305.14160 (2023)."},{"key":"e_1_3_2_1_60_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"Mitigating Hallucinations in Large Vision-Language Models with Instruction Contrastive Decoding. ArXiv abs\/2403.18715","author":"Wang Xintong","year":"2024","unstructured":"Xintong Wang, Jingheng Pan, Liang Ding, and Christian Biemann. 2024. Mitigating Hallucinations in Large Vision-Language Models with Instruction Contrastive Decoding. ArXiv abs\/2403.18715 (2024)."},{"key":"e_1_3_2_1_62_1","volume-title":"CREAM: Consistency Regularized Self-Rewarding Language Models. ArXiv abs\/2410.12735","author":"Wang Zhaoyang","year":"2024","unstructured":"Zhaoyang Wang, Weilei He, Zhiyuan Liang, Xuchao Zhang, Chetan Bansal, Ying Wei, Weitong Zhang, and Huaxiu Yao. 2024. CREAM: Consistency Regularized Self-Rewarding Language Models. ArXiv abs\/2410.12735 (2024)."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681076"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.775"},{"key":"e_1_3_2_1_65_1","volume-title":"Mitigating Object Hallucination via Concentric Causal Attention. arXiv preprint arXiv:2410.15926","author":"Xing Yun","year":"2024","unstructured":"Yun Xing, Yiheng Li, Ivan Laptev, and Shijian Lu. 2024. Mitigating Object Hallucination via Concentric Causal Attention. arXiv preprint arXiv:2410.15926 (2024)."},{"key":"e_1_3_2_1_66_1","volume-title":"LLaVA-Critic: Learning to Evaluate Multimodal Models. ArXiv abs\/2410.02712","author":"Xiong Tianyi","year":"2024","unstructured":"Tianyi Xiong, Xiyao Wang, Dong Guo, Qinghao Ye, Haoqi Fan, Quanquan Gu, Heng Huang, and Chunyuan Li. 2024. LLaVA-Critic: Learning to Evaluate Multimodal Models. ArXiv abs\/2410.02712 (2024)."},{"key":"e_1_3_2_1_67_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 13040-13051","author":"Ye Qinghao","year":"2024","unstructured":"Qinghao Ye, Haiyang Xu, Jiabo Ye, Ming Yan, Anwen Hu, Haowei Liu, Qi Qian, Ji Zhang, and Fei Huang. 2024. mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 13040-13051."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"Hao Yin Guangzong Si and ZileiWang. 2025. ClearSight: Visual Signal Enhancement for Object Hallucination Mitigation in Multimodal Large language Models. (2025).","DOI":"10.1109\/CVPR52734.2025.01363"},{"key":"e_1_3_2_1_69_1","volume-title":"A survey on multimodal large language models. arXiv preprint arXiv:2306.13549","author":"Yin Shukang","year":"2023","unstructured":"Shukang Yin, Chaoyou Fu, Sirui Zhao, Ke Li, Xing Sun, Tong Xu, and Enhong Chen. 2023. A survey on multimodal large language models. arXiv preprint arXiv:2306.13549 (2023)."},{"key":"e_1_3_2_1_70_1","volume-title":"Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback. arXiv preprint arXiv:2312.00849","author":"Yu Tianyu","year":"2023","unstructured":"Tianyu Yu, Yuan Yao, Haoye Zhang, Taiwen He, Yifeng Han, Ganqu Cui, Jinyi Hu, Zhiyuan Liu, Hai-Tao Zheng, Maosong Sun, et al. 2023. Rlhf-v: Towards trustworthy mllms via behavior alignment from fine-grained correctional human feedback. arXiv preprint arXiv:2312.00849 (2023)."},{"key":"e_1_3_2_1_71_1","volume-title":"Mm-vet: Evaluating large multimodal models for integrated capabilities. arXiv preprint arXiv:2308.02490","author":"Yu Weihao","year":"2023","unstructured":"Weihao Yu, Zhengyuan Yang, Linjie Li, Jianfeng Wang, Kevin Lin, Zicheng Liu, Xinchao Wang, and Lijuan Wang. 2023. Mm-vet: Evaluating large multimodal models for integrated capabilities. arXiv preprint arXiv:2308.02490 (2023)."},{"key":"e_1_3_2_1_72_1","volume-title":"Unveiling and Harnessing Hidden Attention Sinks: Enhancing Large Language Models without Training through Attention Calibration. arXiv preprint arXiv:2406.15765","author":"Yu Zhongzhi","year":"2024","unstructured":"Zhongzhi Yu, Zheng Wang, Yonggan Fu, Huihong Shi, Khalid Shaikh, and Yingyan Celine Lin. 2024. Unveiling and Harnessing Hidden Attention Sinks: Enhancing Large Language Models without Training through Attention Calibration. arXiv preprint arXiv:2406.15765 (2024)."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.633"},{"key":"e_1_3_2_1_74_1","volume-title":"Video-llama: An instructiontuned audio-visual language model for video understanding. arXiv preprint arXiv:2306.02858","author":"Zhang Hang","year":"2023","unstructured":"Hang Zhang, Xin Li, and Lidong Bing. 2023. Video-llama: An instructiontuned audio-visual language model for video understanding. arXiv preprint arXiv:2306.02858 (2023)."},{"key":"e_1_3_2_1_75_1","volume-title":"Reflective Instruction Tuning: Mitigating Hallucinations in Large Vision-Language Models. ArXiv abs\/2407.11422","author":"Zhang Jinrui","year":"2024","unstructured":"Jinrui Zhang, Teng Wang, Haigang Zhang, Ping Lu, and Feng Zheng. 2024. Reflective Instruction Tuning: Mitigating Hallucinations in Large Vision-Language Models. ArXiv abs\/2407.11422 (2024)."},{"key":"e_1_3_2_1_76_1","volume-title":"Seeing Clearly by Layer Two: Enhancing Attention Heads to Alleviate Hallucination in LVLMs. ArXiv abs\/2411.09968","author":"Zhang Xiaofeng","year":"2024","unstructured":"Xiaofeng Zhang, Yihao Quan, Chaochen Gu, Chen Shen, Xiaosong Yuan, Shaotian Yan, Hao Cheng, Kaijie Wu, and Jieping Ye. 2024. Seeing Clearly by Layer Two: Enhancing Attention Heads to Alleviate Hallucination in LVLMs. ArXiv abs\/2411.09968 (2024)."},{"key":"e_1_3_2_1_77_1","volume-title":"From Redundancy to Relevance: Enhancing Explainability in Multimodal Large Language Models. arXiv preprint arXiv:2406.06579","author":"Zhang Xiaofeng","year":"2024","unstructured":"Xiaofeng Zhang, Chen Shen, Xiaosong Yuan, Shaotian Yan, Liang Xie, Wenxiao Wang, Chaochen Gu, Hao Tang, and Jieping Ye. 2024. From Redundancy to Relevance: Enhancing Explainability in Multimodal Large Language Models. arXiv preprint arXiv:2406.06579 (2024)."},{"key":"e_1_3_2_1_78_1","volume-title":"Simignore: Exploring and enhancing multimodal large model complex reasoning via similarity computation. Neural Networks","author":"Zhang Xiaofeng","year":"2024","unstructured":"Xiaofeng Zhang, Fanshuo Zeng, and Chaochen Gu. 2024. Simignore: Exploring and enhancing multimodal large model complex reasoning via similarity computation. Neural Networks (2024), 107059."},{"key":"e_1_3_2_1_79_1","volume-title":"Enhancing Multimodal Large Language Models Complex Reason via Similarity Computation. AAAI","author":"Zhang Xiaofeng","year":"2025","unstructured":"Xiaofeng Zhang, Fanshuo Zeng, Yihao Quan, Zheng Hui, and Jiawei Yao. 2025. Enhancing Multimodal Large Language Models Complex Reason via Similarity Computation. AAAI (2025)."},{"key":"e_1_3_2_1_80_1","volume-title":"Few-shot cross-modal text detection via CLIP. The Visual Computer","author":"Zhao Qiyan","year":"2025","unstructured":"Qiyan Zhao, Lanying Liang, Xiaofeng Zhang, Tiange Zhang, Jiuze Li, and Yuefeng Liu. 2025. Few-shot cross-modal text detection via CLIP. The Visual Computer (2025)."},{"key":"e_1_3_2_1_81_1","volume-title":"Analyzing and mitigating object hallucination in large vision-language models. arXiv preprint arXiv:2310.00754","author":"Zhou Yiyang","year":"2023","unstructured":"Yiyang Zhou, Chenhang Cui, Jaehong Yoon, Linjun Zhang, Zhun Deng, Chelsea Finn, Mohit Bansal, and Huaxiu Yao. 2023. Analyzing and mitigating object hallucination in large vision-language models. arXiv preprint arXiv:2310.00754 (2023)."},{"key":"e_1_3_2_1_82_1","volume-title":"Calibrated Self-Rewarding Vision Language Models. ArXiv abs\/2405.14622","author":"Zhou Yiyang","year":"2024","unstructured":"Yiyang Zhou, Zhiyuan Fan, Dongjie Cheng, Sihan Yang, Zhaorun Chen, Chenhang Cui, Xiyao Wang, Yun Li, Linjun Zhang, and Huaxiu Yao. 2024. Calibrated Self-Rewarding Vision Language Models. ArXiv abs\/2405.14622 (2024)."},{"key":"e_1_3_2_1_83_1","volume-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)."},{"key":"e_1_3_2_1_84_1","volume-title":"IBD: Alleviating Hallucinations in Large Vision-Language Models via Image-Biased Decoding. ArXiv abs\/2402.18476","author":"Zhu Lanyun","year":"2024","unstructured":"Lanyun Zhu, Deyi Ji, Tianrun Chen, Peng Xu, Jieping Ye, and Jun Liu. 2024. IBD: Alleviating Hallucinations in Large Vision-Language Models via Image-Biased Decoding. ArXiv abs\/2402.18476 (2024)."},{"key":"e_1_3_2_1_85_1","volume-title":"Unraveling Cross-Modality Knowledge Conflicts in Large Vision-Language Models. ArXiv abs\/2410.03659","author":"Zhu Tinghui","year":"2024","unstructured":"Tinghui Zhu, Qin Liu, Fei Wang, Zhengzhong Tu, and Muhao Chen. 2024. Unraveling Cross-Modality Knowledge Conflicts in Large Vision-Language Models. ArXiv abs\/2410.03659 (2024)."},{"key":"e_1_3_2_1_86_1","volume-title":"Mitigating Object Hallucinations in Large Vision-Language Models via Attention Calibration. ArXiv abs\/2502.01969","author":"Zhu Younan","year":"2025","unstructured":"Younan Zhu, Linwei Tao, Minjing Dong, and Chang Xu. 2025. Mitigating Object Hallucinations in Large Vision-Language Models via Attention Calibration. ArXiv abs\/2502.01969 (2025)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755271","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:44:49Z","timestamp":1765309489000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755271"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":86,"alternative-id":["10.1145\/3746027.3755271","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755271","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}