{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T17:56:08Z","timestamp":1771955768247,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3762014","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:55:00Z","timestamp":1761375300000},"page":"13881-13887","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["HKD4VLM: A Progressive Hybrid Knowledge Distillation Framework for Robust Multimodal Hallucination and Factuality Detection in VLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1748-4924","authenticated-orcid":false,"given":"Zijian","family":"Zhang","sequence":"first","affiliation":[{"name":"Meituan-M17, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6244-0269","authenticated-orcid":false,"given":"Xuecheng","family":"Wu","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5929-6455","authenticated-orcid":false,"given":"Danlei","family":"Huang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, Shaanxi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4701-3054","authenticated-orcid":false,"given":"Siyu","family":"Yan","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6070-3600","authenticated-orcid":false,"given":"Chong","family":"Peng","sequence":"additional","affiliation":[{"name":"Meituan-M17, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7044-1341","authenticated-orcid":false,"given":"Xuezhi","family":"Cao","sequence":"additional","affiliation":[{"name":"Meituan-M17, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katherine Millican Malcolm Reynolds et al. 2022. Flamingo: a visual language model for few-shot learning. Advances in neural information processing systems Vol. 35 (2022) 23716-23736."},{"key":"e_1_3_2_1_2_1","volume-title":"Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-vl: A frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2.5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Lrq-fact: Llm-generated relevant questions for multimodal fact-checking. arXiv preprint arXiv:2410.04616","author":"Beigi Alimohammad","year":"2024","unstructured":"Alimohammad Beigi, Bohan Jiang, Dawei Li, Tharindu Kumarage, Zhen Tan, Pouya Shaeri, and Huan Liu. 2024. Lrq-fact: Llm-generated relevant questions for multimodal fact-checking. arXiv preprint arXiv:2410.04616 (2024)."},{"key":"e_1_3_2_1_5_1","volume-title":"A review of multi-modal large language and vision models. arXiv preprint arXiv:2404.01322","author":"Carolan Kilian","year":"2024","unstructured":"Kilian Carolan, Laura Fennelly, and Alan F Smeaton. 2024. A review of multi-modal large language and vision models. arXiv preprint arXiv:2404.01322 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Multimodal Fact-Checking with Vision Language Models: A Probing Classifier based Solution with Embedding Strategies. arXiv preprint arXiv:2412.05155","author":"Cekinel Recep Firat","year":"2024","unstructured":"Recep Firat Cekinel, Pinar Karagoz, and Cagri Coltekin. 2024. Multimodal Fact-Checking with Vision Language Models: A Probing Classifier based Solution with Embedding Strategies. arXiv preprint arXiv:2412.05155 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Yupeng Chang Xu Wang Jindong Wang Yuan Wu Linyi Yang Kaijie Zhu Hao Chen Xiaoyuan Yi Cunxiang Wang Yidong Wang et al. 2024. A survey on evaluation of large language models. ACM transactions on intelligent systems and technology Vol. 15 3 (2024) 1-45.","DOI":"10.1145\/3641289"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, et al., 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_1_9_1","volume-title":"Junqi Zhao, Weisheng Wang, et al.","author":"Dai Wenliang","year":"2023","unstructured":"Wenliang Dai, Junnan Li, Dongxu Li, Anthony Meng Huat Tiong, Junqi Zhao, Weisheng Wang, et al., 2023. Instructblip: Towards general-purpose vision-language models with instruction tuning. arXiv preprint:2305.06500 (2023)."},{"key":"e_1_3_2_1_10_1","volume-title":"Corey Lynch, Aakanksha Chowdhery, Brian Ichter, Ayzaan Wahid, Jonathan Tompson, Quan Vuong, Tianhe Yu, et al.","author":"Driess Danny","year":"2023","unstructured":"Danny Driess, Fei Xia, Mehdi SM Sajjadi, Corey Lynch, Aakanksha Chowdhery, Brian Ichter, Ayzaan Wahid, Jonathan Tompson, Quan Vuong, Tianhe Yu, et al., 2023. Palm-e: An embodied multimodal language model. arXiv preprint arXiv:2303.03378 (2023)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1213"},{"key":"e_1_3_2_1_12_1","volume-title":"Synergizing rag and reasoning: A systematic review. arXiv preprint arXiv:2504.15909","author":"Gao Yunfan","year":"2025","unstructured":"Yunfan Gao, Yun Xiong, Yijie Zhong, Yuxi Bi, Ming Xue, and Haofen Wang. 2025. Synergizing rag and reasoning: A systematic review. arXiv preprint arXiv:2504.15909 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Large language model based multi-agents: A survey of progress and challenges. arXiv preprint arXiv:2402.01680","author":"Guo Taicheng","year":"2024","unstructured":"Taicheng Guo, Xiuying Chen, Yaqi Wang, Ruidi Chang, Shichao Pei, Nitesh V Chawla, Olaf Wiest, and Xiangliang Zhang. 2024. Large language model based multi-agents: A survey of progress and challenges. arXiv preprint arXiv:2402.01680 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"A comprehensive survey of retrieval-augmented generation (rag): Evolution, current landscape and future directions. arXiv preprint arXiv:2410.12837","author":"Gupta Shailja","year":"2024","unstructured":"Shailja Gupta, Rajesh Ranjan, and Surya Narayan Singh. 2024. A comprehensive survey of retrieval-augmented generation (rag): Evolution, current landscape and future directions. arXiv preprint arXiv:2410.12837 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)."},{"key":"e_1_3_2_1_16_1","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et al., 2022. Lora: Low-rank adaptation of large language models. ICLR, Vol. 1, 2 (2022), 3.","journal-title":"ICLR"},{"key":"e_1_3_2_1_17_1","volume-title":"Andrea Madotto, and Pascale Fung.","author":"Ji Ziwei","year":"2023","unstructured":"Ziwei Ji, Nayeon Lee, Rita Frieske, Tiezheng Yu, Dan Su, Yan Xu, Etsuko Ishii, Ye Jin Bang, Andrea Madotto, and Pascale Fung. 2023. Survey of hallucination in natural language generation. ACM computing surveys, Vol. 55, 12 (2023), 1-38."},{"key":"e_1_3_2_1_18_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Lauren\u00e7on Hugo","year":"2024","unstructured":"Hugo Lauren\u00e7on, Lucile Saulnier, et al., 2024. Obelics: An open web-scale filtered dataset of interleaved image-text documents. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"Wayne Xin Zhao, and Ji-Rong Wen","author":"Li Yifan","year":"2023","unstructured":"Yifan Li, Yifan Du, Kun Zhou, Jinpeng Wang, Wayne Xin Zhao, and Ji-Rong Wen. 2023. Evaluating object hallucination in large vision-language models. arXiv preprint arXiv:2305.10355 (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"Mitigating hallucination in large multi-modal models via robust instruction tuning. arXiv preprint arXiv:2306.14565","author":"Liu Fuxiao","year":"2023","unstructured":"Fuxiao Liu, Kevin Lin, Linjie Li, Jianfeng Wang, Yaser Yacoob, and Lijuan Wang. 2023. Mitigating hallucination in large multi-modal models via robust instruction tuning. arXiv preprint arXiv:2306.14565 (2023)."},{"key":"e_1_3_2_1_21_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Comparative Reasoning for Knowledge Graph Fact Checking. In 2022 IEEE International Conference on Big Data (Big Data).","author":"Liu Lihui","year":"2022","unstructured":"Lihui Liu, Houxiang Ji, Jiejun Xu, and Hanghang Tong. 2022. Comparative Reasoning for Knowledge Graph Fact Checking. In 2022 IEEE International Conference on Big Data (Big Data)."},{"key":"e_1_3_2_1_23_1","unstructured":"Ye Liu Jiajun Zhu Xukai Liu Haoyu Tang Yanghai Zhang Kai Zhang Xiaofang Zhou and Enhong Chen. 2025. Detect Investigate Judge and Determine: A Knowledge-guided Framework for Few-shot Fake News Detection."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Holy Lovenia Wenliang Dai et al. 2023. Negative object presence evaluation (nope) to measure object hallucination in vision-language models. arXiv preprint arXiv:2310.05338 (2023).","DOI":"10.18653\/v1\/2024.alvr-1.4"},{"key":"e_1_3_2_1_25_1","volume-title":"Asad Ullah Khan, et al","author":"Naveed Humza","year":"2023","unstructured":"Humza Naveed, Asad Ullah Khan, et al., 2023. A comprehensive overview of large language models. arXiv preprint arXiv:2307.06435 (2023)."},{"key":"e_1_3_2_1_26_1","unstructured":"OpenAI. 2024. Hello GPT-4o. https:\/\/openai.com\/index\/hello-gpt-4o\/."},{"key":"e_1_3_2_1_27_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703","author":"Paszke A","year":"2019","unstructured":"A Paszke. 2019. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703 (2019)."},{"key":"e_1_3_2_1_28_1","volume-title":"A Deep Learning Approach for Automatic Detection of Fake News. arXiv preprint arXiv:2005.04938","author":"Saikh Tanik","year":"2020","unstructured":"Tanik Saikh, Arkadipta De, Asif Ekbal, and Pushpak Bhattacharyya. 2020. A Deep Learning Approach for Automatic Detection of Fake News. arXiv preprint arXiv:2005.04938 (2020)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.499"},{"key":"e_1_3_2_1_30_1","volume-title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools. arXiv preprint arXiv:2406.12793","author":"Team GLM","year":"2024","unstructured":"GLM Team, Aohan Zeng, Bin Xu, Bowen Wang, Chenhui Zhang, Da Yin, Dan Zhang, Diego Rojas, Guanyu Feng, Hanlin Zhao, et al., 2024. Chatglm: A family of large language models from glm-130b to glm-4 all tools. arXiv preprint arXiv:2406.12793 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"Qvq: To see the world with wisdom.","author":"Team Qwen","year":"2024","unstructured":"Qwen Team. 2024. Qvq: To see the world with wisdom."},{"key":"e_1_3_2_1_32_1","unstructured":"Tencent. 2025. Hunyuan-Vision. https:\/\/hunyuan.tencent.com."},{"key":"e_1_3_2_1_33_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_34_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024a. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_35_1","volume-title":"Efficientvlm: Fast and accurate vision-language models via knowledge distillation and modal-adaptive pruning. arXiv preprint arXiv:2210.07795","author":"Wang Tiannan","year":"2022","unstructured":"Tiannan Wang, Wangchunshu Zhou, Yan Zeng, and Xinsong Zhang. 2022. Efficientvlm: Fast and accurate vision-language models via knowledge distillation and modal-adaptive pruning. arXiv preprint arXiv:2210.07795 (2022)."},{"key":"e_1_3_2_1_36_1","volume-title":"Cogvlm: Visual expert for pretrained language models. arXiv preprint arXiv:2311.03079","author":"Wang Weihan","year":"2023","unstructured":"Weihan Wang, Qingsong Lv, Wenmeng Yu, Wenyi Hong, Ji Qi, Yan Wang, Junhui Ji, Zhuoyi Yang, Lei Zhao, Xixuan Song, et al., 2023. Cogvlm: Visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688985"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00160"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"e_1_3_2_1_40_1","volume-title":"ViC-Bench: Benchmarking Visual-Interleaved Chain-of-Thought Capability in MLLMs with Free-Style Intermediate State Representations. arXiv preprint arXiv:2505.14404","author":"Wu Xuecheng","year":"2025","unstructured":"Xuecheng Wu, Jiaxing Liu, Danlei Huang, Xiaoyu Li, Yifan Wang, Chen Chen, Liya Ma, Xuezhi Cao, and Junxiao Xue. 2025a. ViC-Bench: Benchmarking Visual-Interleaved Chain-of-Thought Capability in MLLMs with Free-Style Intermediate State Representations. arXiv preprint arXiv:2505.14404 (2025)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00854"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3731715.3733453"},{"key":"e_1_3_2_1_43_1","volume-title":"Yu","author":"Yang Yang","year":"2018","unstructured":"Yang Yang, Lei Zheng, Jiawei Zhang, Qingcai Cui, Zhoujun Li, and Philip S. Yu. 2018. TI-CNN: Convolutional Neural Networks for Fake News Detection. arXiv preprint arXiv:1806.00749 (2018)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Yu-Chu Yu Chi-Pin Huang Jr-Jen Chen Kai-Po Chang Yung-Hsuan Lai Fu-En Yang and Yu-Chiang Frank Wang. 2024. Select and Distill: Selective Dual-Teacher Knowledge Transfer for Continual Learning on Vision-Language Models.","DOI":"10.1007\/978-3-031-73347-5_13"},{"key":"e_1_3_2_1_45_1","volume-title":"Deep Mutual Learning. arXiv preprint arXiv:1706.00384","author":"Zhang Ying","year":"2017","unstructured":"Ying Zhang, Tao Xiang, Timothy M. Hospedales, and Huchuan Lu. 2017. Deep Mutual Learning. arXiv preprint arXiv:1706.00384 (2017)."},{"key":"e_1_3_2_1_46_1","volume-title":"VLM-KD: Knowledge Distillation from VLM for Long-Tail Visual Recognition. arXiv preprint arXiv:2408.16930","author":"Zhang Zaiwei","year":"2024","unstructured":"Zaiwei Zhang, Gregory P Meyer, Zhichao Lu, Ashish Shrivastava, Avinash Ravichandran, and Eric M Wolff. 2024. VLM-KD: Knowledge Distillation from VLM for Long-Tail Visual Recognition. arXiv preprint arXiv:2408.16930 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW67362.2025.00120"},{"key":"e_1_3_2_1_48_1","unstructured":"Yuze Zhao Jintao Huang Jinghan Hu Xingjun Wang Yunlin Mao Daoze Zhang Zeyinzi Jiang Zhikai Wu Baole Ai Ang Wang Wenmeng Zhou and Yingda Chen. 2024. SWIFT:A Scalable lightWeight Infrastructure for Fine-Tuning. arXiv:2408.05517 [cs.CL] https:\/\/arxiv.org\/abs\/2408.05517"},{"key":"e_1_3_2_1_49_1","volume-title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. arXiv preprint arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. arXiv preprint arXiv:2304.10592 (2023)."},{"key":"e_1_3_2_1_50_1","unstructured":"Jinguo Zhu Weiyun Wang Zhe Chen Zhaoyang Liu Shenglong Ye Lixin Gu Yuchen Duan Hao Tian Weijie Su Jie Shao et al. 2025. InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models. arXiv preprint arXiv:2504.10479 (2025)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3762014","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T03:59:50Z","timestamp":1765339190000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3762014"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":50,"alternative-id":["10.1145\/3746027.3762014","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3762014","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}