{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:46:31Z","timestamp":1784216791925,"version":"3.55.0"},"reference-count":239,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T00:00:00Z","timestamp":1771632000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T00:00:00Z","timestamp":1771632000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100005090","name":"Beijing Nova Program","doi-asserted-by":"publisher","award":["20230484368"],"award-info":[{"award-number":["20230484368"]}],"id":[{"id":"10.13039\/501100005090","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Strategic Priority Research Program of the Chinese Academy of Sciences","award":["No. XDB0680202"],"award-info":[{"award-number":["No. XDB0680202"]}]},{"name":"Technology Research Project","award":["No. SYG202325"],"award-info":[{"award-number":["No. SYG202325"]}]},{"DOI":"10.13039\/501100004739","name":"Youth Innovation Promotion Association of the Chinese Academy of Sciences","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004739","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s11263-026-02756-9","type":"journal-article","created":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T07:39:28Z","timestamp":1771659568000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["A Survey of Multimodal Hallucination Evaluation and Detection"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6328-2744","authenticated-orcid":false,"given":"Zhiyuan","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0696-2468","authenticated-orcid":false,"given":"Yuecong","family":"Min","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8899-3996","authenticated-orcid":false,"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7948-7033","authenticated-orcid":false,"given":"Bei","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6657-6403","authenticated-orcid":false,"given":"Jiahao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1960-6207","authenticated-orcid":false,"given":"Xiaozhen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8348-392X","authenticated-orcid":false,"given":"Shiguang","family":"Shan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"2756_CR1","unstructured":"Open AI: Hello GPT-4o (2024). https:\/\/openai.com\/zh-Hans-CN\/index\/hello-gpt-4o\/. Accessed 13 Jul 2025."},{"key":"2756_CR2","unstructured":"Team, G., Anil, R., Borgeaud, S., Alayrac, J.\u00a0-B., Yu, J., Soricut, R., Schalkwyk, J., Dai, A.\u00a0M., Hauth, A., Millican, K., et al. (2023). Gemini: A family of highly capable multimodal models. CoRR. arXiv:2312.11805"},{"key":"2756_CR3","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., Ge, W., et al. (2024). Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution. CoRR. arXiv:2409.12191"},{"key":"2756_CR4","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. CVPR, (pp. 10684\u201310695).","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2756_CR5","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., & Sutskever, I. (2021). Zero-shot text-to-image generation. ICML, (pp. 8821\u20138831). Pmlr."},{"key":"2756_CR6","doi-asserted-by":"crossref","unstructured":"Zhang, H., Shao, W., Liu, H., Ma, Y., Luo, P., Qiao, Y., Zheng, N., & Zhang, K. (2024). B-avibench: Towards evaluating the robustness of large vision-language model on black-box adversarial visual-instructions. TIFS.","DOI":"10.1109\/TIFS.2024.3520306"},{"key":"2756_CR7","doi-asserted-by":"crossref","unstructured":"Guo, Q., Pang, S., Jia, X., Liu, Y., & Guo, Q. (2024). Efficient generation of targeted and transferable adversarial examples for vision-language models via diffusion models. TIFS.","DOI":"10.1109\/TIFS.2024.3518072"},{"key":"2756_CR8","doi-asserted-by":"publisher","first-page":"626","DOI":"10.1109\/TIFS.2022.3226905","volume":"18","author":"N Aafaq","year":"2022","unstructured":"Aafaq, N., Akhtar, N., Liu, W., Shah, M., & Mian, A. (2022). Language model agnostic gray-box adversarial attack on image captioning. Transactions on Information Forensics and Security, 18, 626\u2013638.","journal-title":"Transactions on Information Forensics and Security"},{"key":"2756_CR9","doi-asserted-by":"crossref","unstructured":"Zhang, H., Tang, H., Sun, Y., He, S., & Li, Z. (2025). Modality-specific interactive attack for vision-language pre-training models. TIFS.","DOI":"10.1109\/TIFS.2025.3574976"},{"key":"2756_CR10","doi-asserted-by":"publisher","first-page":"5655","DOI":"10.1109\/TIFS.2024.3402179","volume":"19","author":"W Fan","year":"2024","unstructured":"Fan, W., Li, H., Jiang, W., Hao, M., Yu, S., & Zhang, X. (2024). Stealthy targeted backdoor attacks against image captioning. Transactions on Information Forensics and Security, 19, 5655\u20135667.","journal-title":"Transactions on Information Forensics and Security"},{"key":"2756_CR11","first-page":"78723","volume":"36","author":"K Huang","year":"2023","unstructured":"Huang, K., Sun, K., Xie, E., Li, Z., & Liu, X. (2023). T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation. NeurIPS, 36, 78723\u201378747.","journal-title":"NeurIPS"},{"key":"2756_CR12","unstructured":"Bai, Z., Wang, P., Xiao, T., He, T., Han, Z., Zhang, Z., & Shou, M.\u00a0Z. (2024). Hallucination of multimodal large language models: A survey. CoRR. arXiv:2404.18930"},{"key":"2756_CR13","unstructured":"Lan, W., Chen, W., Chen, Q., Pan, S., Zhou, H., & Pan, Y. (2024). A survey of hallucination in large visual language models. CoRR. arXiv:2410.15359"},{"key":"2756_CR14","unstructured":"Liu, H., Xue, W., Chen, Y., Chen, D., Zhao, X., Wang, K., Hou, L., Li, R., & Peng, W. (2024). A survey on hallucination in large vision-language models. CoRR. arXiv:2402.00253"},{"key":"2756_CR15","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Hendricks, L.\u00a0A., Burns, K., Darrell, T., & Saenko, K. (2018). Object hallucination in image captioning. EMNLP, (pp. 4035\u20134045).","DOI":"10.18653\/v1\/D18-1437"},{"key":"2756_CR16","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, W.\u00a0X., Wen, J.\u00a0-R. (2023). Evaluating object hallucination in large vision-language models. EMNLP.","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"2756_CR17","doi-asserted-by":"crossref","unstructured":"Chen, X., Wang, C., Xue, Y., Zhang, N., Yang, X., Li, Q., Shen, Y., Liang, L., Gu, J., & Chen, H. (2024). Unified hallucination detection for multimodal large language models. ACL.","DOI":"10.18653\/v1\/2024.acl-long.178"},{"key":"2756_CR18","doi-asserted-by":"crossref","unstructured":"Hu, Y., Liu, B., Kasai, J., Wang, Y., Ostendorf, M., Krishna, R., & Smith, N.\u00a0A. (2023). Tifa: Accurate and interpretable text-to-image faithfulness evaluation with question answering. ICCV, (pp. 20349\u201320360).","DOI":"10.1109\/ICCV51070.2023.01866"},{"key":"2756_CR19","unstructured":"Wu, M.\u00a0-K., Ji, J., Huang, O., Li, J., Wu, Y., Sun, X., & Ji, R. (2024). Evaluating and analyzing relationship hallucinations in large vision-language models. ICML."},{"key":"2756_CR20","unstructured":"Fu, C., Chen, P., Shen, Y., Qin, Y., Zhang, M., Lin, X., Yang, J., Zheng, X., Li, K., Sun, X., et al. (2023). Mme: A comprehensive evaluation benchmark for multimodal large language models. CoRR. arXiv:2306.13394"},{"key":"2756_CR21","unstructured":"Seth, A., Manocha, D., & Agarwal, C. (2024). Hallucinogen: A benchmark for evaluating object hallucination in large visual-language models. CoRR arXiv:2412.20622"},{"key":"2756_CR22","unstructured":"Zhou, K., Zhu, Y., Chen, Z., Chen, W., Zhao, W.\u00a0X., Chen, X., Lin, Y., Wen, J.\u00a0-R., & Han, J. (2023). Don\u2019t make your llm an evaluation benchmark cheater. CoRR. arXiv:2311.01964"},{"key":"2756_CR23","unstructured":"Feng, W., He, X., Fu, T.\u00a0-J., Jampani, V., Akula, A., Narayana, P., Basu, S., Wang, X.\u00a0E., & Wang, W.\u00a0Y. (2023). Training-free structured diffusion guidance for compositional text-to-image synthesis. ICLR."},{"key":"2756_CR24","unstructured":"Gokhale, T., Palangi, H., Nushi, B., Vineet, V., Horvitz, E., Kamar, E., Baral, C., & Yang, Y. (2022). Benchmarking spatial relationships in text-to-image generation. CoRR. arXiv:2212.10015"},{"key":"2756_CR25","doi-asserted-by":"crossref","unstructured":"Huang, Z., He, W., Long, Q., Wang, Y., Li, H., Yu, Z., Shu, F., Chan, L., Jiang, H., Gan, L., et al. (2024). T2i-factualbench: Benchmarking the factuality of text-to-image models with knowledge-intensive concepts. CoRR. arXiv:2412.04300","DOI":"10.18653\/v1\/2025.acl-long.1334"},{"key":"2756_CR26","doi-asserted-by":"crossref","unstructured":"Danish, S., Sadeghi-Niaraki, A., Khan, S.\u00a0U., Dang, L.\u00a0M., Tightiz, L., & Moon, H. (2025). A comprehensive survey of vision-language models: Pretrained models, fine-tuning, prompt engineering, adapters, and benchmark datasets. Information Fusion, (pp. 103623).","DOI":"10.1016\/j.inffus.2025.103623"},{"key":"2756_CR27","unstructured":"Shen, H., Zhang, J., Xiong, B., Hu, R., Chen, S., Wan, Z., Wang, X., Zhang, Y., Gong, Z., Bao, G., et al. (2025). Efficient diffusion models: A survey. TMLR."},{"key":"2756_CR28","doi-asserted-by":"crossref","unstructured":"Ma, Z., Zhang, Y., Jia, G., Zhao, L., Ma, Y., Ma, M., Liu, G., Zhang, K., Ding, N., Li, J., et al. (2025). Efficient diffusion models: A comprehensive survey from principles to practices. IEEE TPAMI.","DOI":"10.1109\/TPAMI.2025.3569700"},{"key":"2756_CR29","first-page":"44393","volume":"37","author":"X Chen","year":"2024","unstructured":"Chen, X., Ma, Z., Zhang, X., Xu, S., Qian, S., Yang, J., Fouhey, D., & Chai, J. (2024). Multi-object hallucination in vision language models. NeurIPS, 37, 44393\u201344418.","journal-title":"NeurIPS"},{"key":"2756_CR30","doi-asserted-by":"crossref","unstructured":"Huang, W., Liu, H., Guo, M., & Gong, N. (2024). Visual hallucinations of multi-modal large language models. Findings of the ACL, (pp. 9614\u20139631).","DOI":"10.18653\/v1\/2024.findings-acl.573"},{"key":"2756_CR31","doi-asserted-by":"crossref","unstructured":"Yan, B., Chen, Z., Min, Y., Zhang, J., Wang, J., Wang, X., & Shan, S. (2025). Shale: A scalable benchmark for fine-grained hallucination evaluation in lvlms. ACM MM, (pp. 13442\u201313449).","DOI":"10.1145\/3746027.3758308"},{"key":"2756_CR32","doi-asserted-by":"crossref","unstructured":"Park, E., Kim, M., & Kim, G. (2025). Halloc: Token-level localization of hallucinations for vision language models. CVPR, (pp. 29893\u201329903).","DOI":"10.1109\/CVPR52734.2025.02782"},{"key":"2756_CR33","doi-asserted-by":"crossref","unstructured":"Bakr, E.\u00a0M., Sun, P., Shen, X., Khan, F.\u00a0F., Li, L.\u00a0E., & Elhoseiny, M. (2023). Hrs-bench: Holistic, reliable and scalable benchmark for text-to-image models. ICCV, (pp. 20041\u201320053).","DOI":"10.1109\/ICCV51070.2023.01834"},{"key":"2756_CR34","doi-asserted-by":"crossref","unstructured":"Jiang, C., Ye, W., Dong, M., Jia, H., Xu, H., Yan, M., Zhang, J., & Zhang, S. (2024). Hal-eval: A universal and fine-grained hallucination evaluation framework for large vision language models. ACM MM.","DOI":"10.1145\/3664647.3680576"},{"key":"2756_CR35","doi-asserted-by":"crossref","unstructured":"Ayaz, M., Khan, M., Saqib, M., Khelifi, A., Sajjad, M., & Elsaddik, A. (2024). Medvlm: Medical vision-language model for consumer devices. IEEE CEM.","DOI":"10.1109\/MCE.2024.3522521"},{"key":"2756_CR36","doi-asserted-by":"crossref","unstructured":"Arshad, M.\u00a0A., Jubery, T.\u00a0Z., Roy, T., Nassiri, R., Singh, A.\u00a0K., Singh, A., Hegde, C., Ganapathysubramanian, B., Balu, A., Krishnamurthy, A., et al. (2025). Leveraging vision language models for specialized agricultural tasks. WACV, (pp. 6320\u20136329). IEEE.","DOI":"10.1109\/WACV61041.2025.00616"},{"issue":"25","key":"2756_CR37","first-page":"26290","volume":"39","author":"Y Lim","year":"2025","unstructured":"Lim, Y., Choi, H., & Shim, H. (2025). Evaluating image hallucination in text-to-image generation with question-answering. Association for the Advancement of Artificial Intelligence, 39(25), 26290\u201326298.","journal-title":"Association for the Advancement of Artificial Intelligence"},{"key":"2756_CR38","unstructured":"Hu, H., Zhang, J., Zhao, M., & Sun, Z. (2023). Ciem: Contrastive instruction evaluation method for better instruction tuning. NeurIPS Workshop."},{"key":"2756_CR39","unstructured":"Chen, Z., Zhu, Y., Zhan, Y., Li, Z., Zhao, C., Wang, J., & Tang, M. (2023). Mitigating hallucination in visual language models with visual supervision. CoRR. arXiv:2311.16479"},{"key":"2756_CR40","doi-asserted-by":"crossref","unstructured":"Guan, T., Liu, F., Wu, X., Xian, R., Li, Z., Liu, X., Wang, X., Chen, L., Huang, F., Yacoob, Y., Manocha, D., & Zhou, T. (2023). Hallusionbench: An advanced diagnostic suite for entangled language hallucination and visual illusion in large vision-language models. CVPR, (pp. 14375\u201314385).","DOI":"10.1109\/CVPR52733.2024.01363"},{"key":"2756_CR41","doi-asserted-by":"crossref","unstructured":"Wang, L., He, J., Li, S., Liu, N., & Lim, E.\u00a0-P. (2023). Mitigating fine-grained hallucination by fine-tuning large vision-language models with caption rewrites. MMM","DOI":"10.1007\/978-3-031-53302-0_3"},{"key":"2756_CR42","unstructured":"Qiu, H., Huang, J., Gao, P., Qi, Q., Zhang, X., Shao, L., & Lu, S. (2024). Longhalqa: Long-context hallucination evaluation for multimodal large language models. CoRR. arXiv:2410.09962"},{"key":"2756_CR43","doi-asserted-by":"crossref","unstructured":"Liu, J., Fu, Y., Xie, R., Xie, R., Sun, X., Lian, F., Kang, Z., & Li, X. (2025). Phd: A chatgpt-prompted visual hallucination evaluation dataset. CVPR, (pp. 19857\u201319866).","DOI":"10.1109\/CVPR52734.2025.01849"},{"key":"2756_CR44","doi-asserted-by":"crossref","unstructured":"Cao, Q., Cheng, J., Liang, X., & Lin, L. (2024). Visdiahalbench: A visual dialogue benchmark for diagnosing hallucination in large vision-language models. ACL, (pp. 12161\u201312176).","DOI":"10.18653\/v1\/2024.acl-long.658"},{"key":"2756_CR45","doi-asserted-by":"crossref","unstructured":"Wu, X., Guan, T., Li, D., Huang, S., Liu, X., Wang, X., Xian, R., Shrivastava, A., Huang, F., Boyd-Graber, J., et al. (2024). Autohallusion: Automatic generation of hallucination benchmarks for vision-language models. Findings of the EMNLP, (pp. 8395\u20138419).","DOI":"10.18653\/v1\/2024.findings-emnlp.493"},{"key":"2756_CR46","doi-asserted-by":"crossref","unstructured":"Ben-Kish, A., Yanuka, M., Alper, M., Giryes, R., & Averbuch-Elor, H. (2024). Mitigating open-vocabulary caption hallucinations. EMNLP, (pp. 22680\u201322698).","DOI":"10.18653\/v1\/2024.emnlp-main.1263"},{"key":"2756_CR47","unstructured":"Zhai, B., Yang, S., Xu, C., Shen, S., Keutzer, K., Li, C., & Li, M. (2023). Halle-control: Controlling object hallucination in large multimodal models. CoRR. arXiv:2310.01779"},{"key":"2756_CR48","doi-asserted-by":"crossref","unstructured":"Sun, Z., Shen, S., Cao, S., Liu, H., Li, C., Shen, Y., Gan, C., Gui, L., Wang, Y.\u00a0-X., Yang, Y., et al. (2024). Aligning large multimodal models with factually augmented rlhf. Findings of the ACL, (pp. 13088\u201313110).","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"2756_CR49","unstructured":"Liu, F., Lin, K., Li, L., Wang, J., Yacoob, Y., & Wang, L. (2023). Mitigating hallucination in large multi-modal models via robust instruction tuning. ICLR."},{"key":"2756_CR50","doi-asserted-by":"crossref","unstructured":"Lovenia, H., Dai, W., Cahyawijaya, S., Ji, Z., & Fung, P. (2024). Negative object presence evaluation (nope) to measure object hallucination in vision-language models. ALVR Workshops, (pp. 37\u201358).","DOI":"10.18653\/v1\/2024.alvr-1.4"},{"key":"2756_CR51","unstructured":"Wang, J., Zhou, Y., Xu, G., Shi, P., Zhao, C., Xu, H., Ye, Q., Yan, M., Zhang, J., Zhu, J., et al. (2023). Evaluation and analysis of hallucination in large vision-language models. CoRR. arXiv:2308.15126"},{"key":"2756_CR52","unstructured":"Rani, A., Rawte, V., Sharma, H., Anand, N., Rajbangshi, K., Sheth, A., & Das, A. (2024). Visual hallucination: Definition, quantification, and prescriptive remediations. CoRR. arXiv:2403.17306"},{"key":"2756_CR53","doi-asserted-by":"crossref","unstructured":"Villa, A., L\u00e9on, J., Soto, A., & Ghanem, B. (2025). Behind the magic, merlim: Multi-modal evaluation benchmark for large image-language models. CVPR, (pp. 492\u2013502).","DOI":"10.1109\/CVPRW67362.2025.00054"},{"key":"2756_CR54","unstructured":"Wang, J., Wang, Y., Xu, G., Zhang, J., Gu, Y., Jia, H., Wang, J., Xu, H., Yan, M., Zhang, J., et al. (2023). Amber: An llm-free multi-dimensional benchmark for mllms hallucination evaluation. CoRR. arXiv:2311.07397"},{"key":"2756_CR55","unstructured":"Chen, J., Yang, D., Wu, T., Jiang, Y., Hou, X., Li, M., Wang, S., Xiao, D., Li, K., & Zhang, L. (2024). Detecting and evaluating medical hallucinations in large vision language models. CoRR. arXiv:2406.10185"},{"key":"2756_CR56","doi-asserted-by":"crossref","unstructured":"Tu, Y., Hu, R., & Sang, J. (2025). Ode: Open-set evaluation of hallucinations in multimodal large language models. CVPR, (pp. 19836\u201319845).","DOI":"10.1109\/CVPR52734.2025.01847"},{"key":"2756_CR57","doi-asserted-by":"crossref","unstructured":"Cho, J., Zala, A., & Bansal, M. (2023). Dall-eval: Probing the reasoning skills and social biases of text-to-image generation models. ICCV, (pp. 3043\u20133054).","DOI":"10.1109\/ICCV51070.2023.00283"},{"key":"2756_CR58","unstructured":"Niu, Y., Ning, M., Zheng, M., Lin, B., Jin, P., Liao, J., Ning, K., Zhu, B., & Yuan, L. (2025). Wise: A world knowledge-informed semantic evaluation for text-to-image generation. CoRR. arXiv:2503.07265"},{"key":"2756_CR59","doi-asserted-by":"crossref","unstructured":"Li, B., Lin, Z., Pathak, D., Li, J., Fei, Y., Wu, K., Xia, X., Zhang, P., Neubig, G., & Ramanan, D. (2024). Evaluating and improving compositional text-to-visual generation. CVPR, (pp. 5290\u20135301).","DOI":"10.1109\/CVPRW63382.2024.00538"},{"key":"2756_CR60","doi-asserted-by":"crossref","unstructured":"Jing, L., Li, R., Chen, Y., & Du, X. (2023). Faithscore: Fine-grained evaluations of hallucinations in large vision-language models. EMNLP.","DOI":"10.18653\/v1\/2024.findings-emnlp.290"},{"key":"2756_CR61","doi-asserted-by":"crossref","unstructured":"Kaul, P., Li, Z., Yang, H., Dukler, Y., Swaminathan, A., Taylor, C., & Soatto, S. (2024). Throne: An object-based hallucination benchmark for the free-form generations of large vision-language models. CVPR, (pp. 27228\u201327238).","DOI":"10.1109\/CVPR52733.2024.02571"},{"key":"2756_CR62","unstructured":"Nguyen, C.\u00a0-D., Wu, X., Vu, D.\u00a0A., Zhao, S., Nguyen, T., & Luu, A.\u00a0T. (2025). Cutpaste &find: Efficient multimodal hallucination detector with visual-aid knowledge base. CoRR. arXiv:2502.12591"},{"key":"2756_CR63","doi-asserted-by":"crossref","unstructured":"Huang, J.\u00a0-H., Zhu, H., Shen, Y., Rudinac, S., & Kanoulas, E. (2025). Image2text2image: A novel framework for label-free evaluation of image-to-text generation with text-to-image diffusion models. MMM, (pp. 413\u2013427). Springer","DOI":"10.1007\/978-981-96-2071-5_30"},{"key":"2756_CR64","doi-asserted-by":"crossref","unstructured":"Gunjal, A., Yin, J., & Bas, E. (2023). Detecting and preventing hallucinations in large vision language models. AAAI.","DOI":"10.1609\/aaai.v38i16.29771"},{"key":"2756_CR65","doi-asserted-by":"crossref","unstructured":"Xiao, W., Huang, Z., Gan, L., He, W., Li, H., Yu, Z., Shu, F., Jiang, H., & Zhu, L. (2025). Detecting and mitigating hallucination in large vision language models via fine-grained ai feedback. AAAI.","DOI":"10.1609\/aaai.v39i24.34744"},{"key":"2756_CR66","doi-asserted-by":"crossref","unstructured":"Heiman, A., Zhang, X., Chen, E., Kim, S.\u00a0E., & Rajpurkar, P. (2025). Factchexcker: Mitigating measurement hallucinations in chest x-ray report generation models. CVPR, (pp. 30787\u201330796).","DOI":"10.1109\/CVPR52734.2025.02867"},{"key":"2756_CR67","unstructured":"Zhang, R., Zhang, H., & Zheng, Z. (2024). Vl-uncertainty: Detecting hallucination in large vision-language model via uncertainty estimation. CoRR arXiv:2411.11919"},{"key":"2756_CR68","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Xie, R., Chen, J., Sun, X., Wang, Y., et al. (2024). Dhcp: Detecting hallucinations by cross-modal attention pattern in large vision-language models. CoRR. arXiv:2411.18659","DOI":"10.1145\/3746027.3755118"},{"key":"2756_CR69","doi-asserted-by":"crossref","unstructured":"Huang, Q., Dong, X., Zhang, P., Wang, B., He, C., Wang, J., Lin, D., Zhang, W., & Yu, N. (2024). Opera: Alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation. CVPR, (pp. 13418\u201313427).","DOI":"10.1109\/CVPR52733.2024.01274"},{"key":"2756_CR70","unstructured":"Jiang, N., Kachinthaya, A., Petryk, S., & Gandelsman, Y. (2025). Interpreting and editing vision-language representations to mitigate hallucinations. ICLR."},{"key":"2756_CR71","doi-asserted-by":"crossref","unstructured":"Phukan, A., Divyansh, D., Morj, H.\u00a0K., Vaishnavi, V., Saxena, A., & Goswami, K. (2025). Beyond logit lens: Contextual embeddings for robust hallucination detection & grounding in vlms. NAACL, (pp. 9661\u2013967)5","DOI":"10.18653\/v1\/2025.naacl-long.488"},{"key":"2756_CR72","doi-asserted-by":"crossref","unstructured":"Cho, J., Zala, A., & Bansal, M. (2022). Dall-eval: Probing the reasoning skills and social biases of text-to-image generation models. ICCV, (pp. 3020\u20133031).","DOI":"10.1109\/ICCV51070.2023.00283"},{"key":"2756_CR73","first-page":"6048","volume":"36","author":"J Cho","year":"2023","unstructured":"Cho, J., Zala, A., & Bansal, M. (2023). Visual programming for step-by-step text-to-image generation and evaluation. NeurIPS, 36, 6048\u20136069.","journal-title":"NeurIPS"},{"key":"2756_CR74","first-page":"23075","volume":"36","author":"Y Lu","year":"2023","unstructured":"Lu, Y., Yang, X., Li, X., Wang, X. E., & Wang, W. Y. (2023). Llmscore: Unveiling the power of large language models in text-to-image synthesis evaluation. NeurIPS, 36, 23075\u201323093.","journal-title":"NeurIPS"},{"key":"2756_CR75","unstructured":"Qin, Z., Cheng, D., Wang, H., Yi, H., Shao, Y., Fan, Z., Li, K., & Lao, Q. (2024). Evaluating hallucination in text-to-image diffusion models with scene-graph based question-answering agent. CoRR. arXiv:2412.05722"},{"key":"2756_CR76","unstructured":"Cho, J., Hu, Y., Baldridge, J.\u00a0M., Garg, R., Anderson, P., Krishna, R., Bansal, M., Pont-Tuset, J., & Wang, S. (2024). Davidsonian scene graph: Improving reliability in fine-grained evaluation for text-to-image generation. ICLR"},{"key":"2756_CR77","doi-asserted-by":"crossref","unstructured":"Lin, Z., Pathak, D., Li, B., Li, J., Xia, X., Neubig, G., Zhang, P., & Ramanan, D. (2024). Evaluating text-to-visual generation with image-to-text generation. ECCV, (pp. 366\u2013384). Springer","DOI":"10.1007\/978-3-031-72673-6_20"},{"key":"2756_CR78","doi-asserted-by":"crossref","unstructured":"Fei, H., Luo, M., Xu, J., Wu, S., Ji, W., Lee, M.\u00a0-L., & Hsu, W. (2024). Fine-grained structural hallucination detection for unified visual comprehension and generation in multimodal llm. ACM MM Workshops, (pp. 13\u201322).","DOI":"10.1145\/3689090.3689388"},{"key":"2756_CR79","first-page":"34892","volume":"36","author":"H Liu","year":"2023","unstructured":"Liu, H., Li, C., Wu, Q., & Lee, Y. J. (2023). Visual instruction tuning. NeurIPS, 36, 34892\u201334916.","journal-title":"NeurIPS"},{"key":"2756_CR80","unstructured":"OpenAI (2023). GPT-4V(ision) System Card. https:\/\/cdn.openai.com\/papers\/GPTV_System_Card.pdf. Accessed 04 Dec 2025"},{"key":"2756_CR81","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., & Sutskever, I. (2021). Learning transferable visual models from natural language supervision. ICML."},{"key":"2756_CR82","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., & Lee, Y.J. (2024). Llava-next: Improved reasoning, ocr, and world knowledge, january 2024. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next1(8)"},{"key":"2756_CR83","first-page":"157","volume":"12","author":"NF Liu","year":"2024","unstructured":"Liu, N. F., Lin, K., Hewitt, J., Paranjape, A., Bevilacqua, M., Petroni, F., & Liang, P. (2024). Lost in the middle: How language models use long contexts. ACL, 12, 157\u2013173.","journal-title":"ACL"},{"key":"2756_CR84","unstructured":"Kang, S., Kim, J., Kim, J., & Hwang, S.\u00a0J. (2025). See what you are told: Visual attention sink in large multimodal models. ICLR"},{"key":"2756_CR85","first-page":"17612","volume":"35","author":"VW Liang","year":"2022","unstructured":"Liang, V. W., Zhang, Y., Kwon, Y., Yeung, S., & Zou, J. Y. (2022). Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning. NeurIPS, 35, 17612\u201317625.","journal-title":"NeurIPS"},{"key":"2756_CR86","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023). Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. ICML, (pp. 19730\u201319742)."},{"key":"2756_CR87","doi-asserted-by":"crossref","unstructured":"Yan, S., Bai, M., Chen, W., Zhou, X., Huang, Q., & Li, L.\u00a0E. (2024). Vigor: Improving visual grounding of large vision language models with fine-grained reward modeling. ECCV, (pp. 37\u201353). Springer.","DOI":"10.1007\/978-3-031-73030-6_3"},{"key":"2756_CR88","first-page":"23716","volume":"35","author":"J-B Alayrac","year":"2022","unstructured":"Alayrac, J.-B., Donahue, J., Luc, P., Miech, A., Barr, I., Hasson, Y., Lenc, K., Mensch, A., Millican, K., Reynolds, M., et al. (2022). Flamingo: A visual language model for few-shot learning. NeurIPS, 35, 23716\u201323736.","journal-title":"NeurIPS"},{"key":"2756_CR89","doi-asserted-by":"crossref","unstructured":"He, J., Zhu, K., Guo, H., Fang, J., Hua, Z., Jia, Y., Tang, M., Chua, T.\u00a0-S., & Wang, J. (2025). Cracking the code of hallucination in lvlms with vision-aware head divergence. ACL, (pp. 3488\u20133501).","DOI":"10.18653\/v1\/2025.acl-long.175"},{"key":"2756_CR90","doi-asserted-by":"crossref","unstructured":"Pi, R., Miao, K., Peihang, L., Liu, R., Gao, J., Zhang, J., & Zhou, X. (2025). Pointing to a llama and call it a camel: On the sycophancy of multimodal large language models. EMNLP, (pp. 20177\u201320191).","DOI":"10.18653\/v1\/2025.emnlp-main.1020"},{"key":"2756_CR91","unstructured":"Zhu, T., Liu, Q., Wang, F., Tu, Z., & Chen, M. (2024). Unraveling cross-modality knowledge conflicts in large vision-language models. CoRR. arXiv:2410.03659"},{"key":"2756_CR92","unstructured":"Zhang, R., Han, J., Liu, C., Gao, P., Zhou, A., Hu, X., Yan, S., Lu, P., Li, H., & Qiao, Y. (2023). Llama-adapter: Efficient fine-tuning of language models with zero-init attention. CoRR. arXiv:2303.16199"},{"key":"2756_CR93","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., & Lee, Y.\u00a0J. (2024). Improved baselines with visual instruction tuning. CVPR, (pp. 26296\u201326306).","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"2756_CR94","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., & Elhoseiny, M. (2024). Minigpt-4: Enhancing vision-language understanding with advanced large language models. ICLR."},{"key":"2756_CR95","unstructured":"Christiano, P.F., Leike, J., Brown, T., Martic, M., Legg, S., & Amodei, D. (2017). Deep reinforcement learning from human preferences. NeurIPS30."},{"key":"2756_CR96","first-page":"53728","volume":"36","author":"R Rafailov","year":"2023","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C. D., Ermon, S., & Finn, C. (2023). Direct preference optimization: Your language model is secretly a reward model. NeurIPS, 36, 53728\u201353741.","journal-title":"NeurIPS"},{"key":"2756_CR97","unstructured":"Zhao, Z., Wang, B., Ouyang, L., Dong, X., Wang, J., & He, C. (2023). Beyond hallucinations: Enhancing lvlms through hallucination-aware direct preference optimization. CoRR. arXiv:2311.16839"},{"key":"2756_CR98","unstructured":"Esser, P., Kulal, S., Blattmann, A., Entezari, R., M\u00fcller, J., Saini, H., Levi, Y., Lorenz, D., Sauer, A., Boesel, F., et al. (2024). Scaling rectified flow transformers for high-resolution image synthesis. ICML, (pp. 12606\u201312633)."},{"key":"2756_CR99","unstructured":"Betker, J., Goh, G., Jing, L., Brooks, T., Wang, J., Li, L., Ouyang, L., Zhuang, J., Lee, J., Guo, Y., et al. (2023). Improving image generation with better captions. Computer Science. 2(3), 8. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf"},{"key":"2756_CR100","doi-asserted-by":"crossref","unstructured":"Peebles, W., & Xie, S. (2023). Scalable diffusion models with transformers. CVPR, (pp. 4195\u20134205).","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"2756_CR101","unstructured":"Koishigarina, D., Uselis, A., & Oh, S.\u00a0J. (2025). Clip behaves like a bag-of-words model cross-modally but not uni-modally. CoRR. arXiv:2502.03566"},{"issue":"70","key":"2756_CR102","first-page":"1","volume":"25","author":"HW Chung","year":"2024","unstructured":"Chung, H. W., Hou, L., Longpre, S., Zoph, B., Tay, Y., Fedus, W., Li, Y., Wang, X., Dehghani, M., Brahma, S., et al. (2024). Scaling instruction-finetuned language models. Journal of Machine Learning Research, 25(70), 1\u201353.","journal-title":"Journal of Machine Learning Research"},{"key":"2756_CR103","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., et al. (2023). Llama 2: Open foundation and fine-tuned chat models. CoRR. arXiv:2307.09288"},{"key":"2756_CR104","unstructured":"Yang, A., Xiao, B., Wang, B., Zhang, B., Bian, C., Yin, C., Lv, C., Pan, D., Wang, D., Yan, D., et al. (2023). Baichuan 2: Open large-scale language models. CoRR. arXiv:2309.10305"},{"key":"2756_CR105","unstructured":"Jiao, Q., Chen, D., Huang, Y., Lin, X., Shen, Y., & Li, Y. (2025). Detailmaster: Can your text-to-image model handle long prompts? CoRR. arXiv:2505.16915"},{"key":"2756_CR106","unstructured":"Stability AI (2024). Stable Diffusion 3. https:\/\/stability.ai\/news\/stable-diffusion-3-research-paper. Accessed 04 Dec 2025."},{"key":"2756_CR107","doi-asserted-by":"crossref","unstructured":"Relic, L., Azevedo, R., Gross, M., & Schroers, C. (2024). Lossy image compression with foundation diffusion models. ECCV, (pp. 303\u2013319). Springer","DOI":"10.1007\/978-3-031-73030-6_17"},{"key":"2756_CR108","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., & Ommer, B. (2021). Taming transformers for high-resolution image synthesis. CVPR, (pp. 12873\u201312883).","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"2756_CR109","unstructured":"Van Den\u00a0Oord, A., & Vinyals, O., (2017). Neural discrete representation learning. NeurIPS30."},{"key":"2756_CR110","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. NeurIPS, 33, 6840\u20136851.","journal-title":"NeurIPS"},{"key":"2756_CR111","unstructured":"Lu, R., Wang, R., Lyu, K., Jiang, X., Huang, G., & Wang, M. (2025). Towards understanding text hallucination of diffusion models via local generation bias. ICLR."},{"key":"2756_CR112","unstructured":"Black Forest Labs (2024). FLUX.1-dev. https:\/\/huggingface.co\/black-forest-labs\/FLUX.1-dev. Accessed 04 Dec 2025"},{"key":"2756_CR113","unstructured":"Liu, X., & Gong, C., (2022). Flow straight and fast: Learning to generate and transfer data with rectified flow. ICLR."},{"key":"2756_CR114","unstructured":"Lipman, Y., Chen, R.\u00a0T., Ben-Hamu, H., Nickel, M., & Le, M. (2023). Flow matching for generative modeling. ICLR."},{"key":"2756_CR115","unstructured":"Hu, E.J., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., Chen, W., et al. (2022). Lora: Low-rank adaptation of large language models. ICLR."},{"key":"2756_CR116","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. ICCV, (pp. 3836\u20133847).","DOI":"10.1109\/ICCV51070.2023.00355"},{"issue":"1","key":"2756_CR117","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1037\/0033-2909.107.1.82","volume":"107","author":"RP Bentall","year":"1990","unstructured":"Bentall, R. P. (1990). The illusion of reality: A review and integration of psychological research on hallucinations. Psychological Bulletin, 107(1), 82.","journal-title":"Psychological Bulletin"},{"key":"2756_CR118","unstructured":"Slade, P. D., & Bentall, R. P. (1988). Sensory Deception: A Scientific Analysis of Hallucination. Johns Hopkins series in contemporary medicine and public health. Johns Hopkins University Press, London."},{"issue":"12","key":"2756_CR119","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3571730","volume":"55","author":"Z Ji","year":"2023","unstructured":"Ji, Z., Lee, N., Frieske, R., Yu, T., Su, D., Xu, Y., Ishii, E., Bang, Y. J., Madotto, A., & Fung, P. (2023). Survey of hallucination in natural language generation. ACM Computing Surveys, 55(12), 1\u201338.","journal-title":"ACM Computing Surveys"},{"key":"2756_CR120","doi-asserted-by":"crossref","unstructured":"Yu, J., Lin, Z., Yang, J., Shen, X., Lu, X., & Huang, T.\u00a0S. (2018). Generative image inpainting with contextual attention. CVPR, (pp. 5505\u20135514).","DOI":"10.1109\/CVPR.2018.00577"},{"key":"2756_CR121","doi-asserted-by":"crossref","unstructured":"Baker, S., & Kanade, T. (2000). Hallucinating faces. IEEE FG, (pp. 83\u201388). IEEE.","DOI":"10.1109\/AFGR.2000.840616"},{"issue":"1","key":"2756_CR122","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1007\/s11263-006-0029-5","volume":"75","author":"C Liu","year":"2007","unstructured":"Liu, C., Shum, H.-Y., & Freeman, W. T. (2007). Face hallucination: Theory and practice. International Journal of Computer Vision, 75(1), 115\u2013134.","journal-title":"International Journal of Computer Vision"},{"issue":"1","key":"2756_CR123","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/s11263-013-0645-9","volume":"106","author":"N Wang","year":"2014","unstructured":"Wang, N., Tao, D., Gao, X., Li, X., & Li, J. (2014). A comprehensive survey to face hallucination. International Journal of Computer Vision, 106(1), 9\u201330.","journal-title":"International Journal of Computer Vision"},{"issue":"6","key":"2756_CR124","doi-asserted-by":"publisher","first-page":"763","DOI":"10.1007\/s11263-019-01154-8","volume":"127","author":"H Huang","year":"2019","unstructured":"Huang, H., He, R., Sun, Z., & Tan, T. (2019). Wavelet domain generative adversarial network for multi-scale face hallucination. International Journal of Computer Vision, 127(6), 763\u2013784.","journal-title":"International Journal of Computer Vision"},{"issue":"7","key":"2756_CR125","doi-asserted-by":"publisher","first-page":"2367","DOI":"10.1007\/s11263-023-01977-6","volume":"132","author":"W Quan","year":"2024","unstructured":"Quan, W., Chen, J., Liu, Y., Yan, D.-M., & Wonka, P. (2024). Deep learning-based image and video inpainting: A survey. International Journal of Computer Vision, 132(7), 2367\u20132400.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR126","doi-asserted-by":"crossref","unstructured":"Hariharan, B., & Girshick, R. (2017). Low-shot visual recognition by shrinking and hallucinating features. ICCV, (pp. 3018\u20133027).","DOI":"10.1109\/ICCV.2017.328"},{"issue":"6","key":"2756_CR127","doi-asserted-by":"publisher","first-page":"3689","DOI":"10.1007\/s11263-025-02357-y","volume":"133","author":"M Yang","year":"2025","unstructured":"Yang, M., & Wang, Z. (2025). Image synthesis under limited data: A survey and taxonomy. International Journal of Computer Vision, 133(6), 3689\u20133726.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR128","doi-asserted-by":"crossref","unstructured":"Kayhan, O.\u00a0S., Vredebregt, B., & Van\u00a0Gemert, J.\u00a0C. (2021). Hallucination in object detection\u2013a study in visual part verification. ICIP, (pp. 2234\u20132238). IEEE.","DOI":"10.1109\/ICIP42928.2021.9506670"},{"issue":"1","key":"2756_CR129","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1007\/s11263-020-01363-6","volume":"129","author":"N Piasco","year":"2021","unstructured":"Piasco, N., Sidib\u00e9, D., Gouet-Brunet, V., & Demonceaux, C. (2021). Improving image description with auxiliary modality for visual localization in challenging conditions. International Journal of Computer Vision, 129(1), 185\u2013202.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR130","unstructured":"You, J., Shi, H., Jiang, Z., Huang, Z., Gan, R., Wu, K., Cheng, X., Li, X., & Ran, B. (2024). V2x-vlm: End-to-end v2x cooperative autonomous driving through large vision-language models. CoRR. arXiv:2408.09251"},{"key":"2756_CR131","first-page":"134614","volume":"37","author":"SK Aithal","year":"2024","unstructured":"Aithal, S. K., Maini, P., Lipton, Z., & Kolter, J. Z. (2024). Understanding hallucinations in diffusion models through mode interpolation. NeurIPS, 37, 134614\u2013134644.","journal-title":"NeurIPS"},{"key":"2756_CR132","doi-asserted-by":"crossref","unstructured":"Kim, S., Jin, C., Diethe, T., Figini, M., Tregidgo, H.\u00a0F., Mullokandov, A., Teare, P., & Alexander, D.\u00a0C. (2024). Tackling structural hallucination in image translation with local diffusion. ECCV, (pp. 87\u2013103). Springer.","DOI":"10.1007\/978-3-031-73004-7_6"},{"key":"2756_CR133","first-page":"118883","volume":"37","author":"H Wang","year":"2024","unstructured":"Wang, H., Cao, J., Liu, J., Zhou, X., Huang, H., & He, R. (2024). Hallo3d: Multi-modal hallucination detection and mitigation for consistent 3d content generation. NeurIPS, 37, 118883\u2013118906.","journal-title":"NeurIPS"},{"key":"2756_CR134","unstructured":"Saleh, M., & Tabatabaei, A. (2025). Building trustworthy multimodal ai: A review of fairness, transparency, and ethics in vision-language tasks. CoRR. arXiv:2504.13199"},{"issue":"4","key":"2756_CR135","first-page":"1345","volume":"10","author":"J Agnese","year":"2020","unstructured":"Agnese, J., Herrera, J., Tao, H., & Zhu, X. (2020). A survey and taxonomy of adversarial neural networks for text-to-image synthesis. Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery, 10(4), 1345.","journal-title":"Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery"},{"issue":"2","key":"2756_CR136","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3703155","volume":"43","author":"L Huang","year":"2025","unstructured":"Huang, L., Yu, W., Ma, W., Zhong, W., Feng, Z., Wang, H., Chen, Q., Peng, W., Feng, X., Qin, B., et al. (2025). A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions. ACM Transactions on Information Systems, 43(2), 1\u201355.","journal-title":"ACM Transactions on Information Systems"},{"key":"2756_CR137","doi-asserted-by":"crossref","unstructured":"Dai, W., Liu, Z., Ji, Z., Su, D., & Fung, P. (2023). Plausible may not be faithful: Probing object hallucination in vision-language pre-training. EACL, (pp. 2136\u20132148).","DOI":"10.18653\/v1\/2023.eacl-main.156"},{"key":"2756_CR138","doi-asserted-by":"crossref","unstructured":"Nesteruk, S., Svetlana, I., & Andrey, S. (2024). Image dataset augmentation a survey and taxonomy. Measurements and Instrumentation for Machine Vision, (pp. 110\u2013136).","DOI":"10.1201\/9781003343783-5"},{"key":"2756_CR139","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. NeurIPS, 35, 36479\u201336494.","journal-title":"NeurIPS"},{"key":"2756_CR140","unstructured":"Meng, F., Shao, W., Luo, L., Wang, Y., Chen, Y., Lu, Q., Yang, Y., Yang, T., Zhang, K., Qiao, Y., et al. (2024). Phybench: A physical commonsense benchmark for evaluating text-to-image models. CoRR. arXiv:2406.11802"},{"key":"2756_CR141","doi-asserted-by":"publisher","first-page":"1430984","DOI":"10.3389\/frai.2024.1430984","volume":"7","author":"I Hartsock","year":"2024","unstructured":"Hartsock, I., & Rasool, G. (2024). Vision-language models for medical report generation and visual question answering: A review. Frontiers in Artificial Intelligence, 7, 1430984.","journal-title":"Frontiers in Artificial Intelligence"},{"key":"2756_CR142","unstructured":"Ye, J., & Tang, H. (2025). Multimodal large language models for medicine: A comprehensive survey. CoRR. arXiv:2504.21051"},{"key":"2756_CR143","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.\u00a0Q., & Artzi, Y. (2020). Bertscore: Evaluating text generation with bert. ICLR."},{"key":"2756_CR144","doi-asserted-by":"crossref","unstructured":"Uzunova, H., Ehrhardt, J., Jacob, F., Frydrychowicz, A., & Handels, H. (2019). Multi-scale gans for memory-efficient generation of high resolution medical images. MICCAI, (pp. 112\u2013120).","DOI":"10.1007\/978-3-030-32226-7_13"},{"key":"2756_CR145","doi-asserted-by":"crossref","unstructured":"Pinaya, W.\u00a0H., Tudosiu, P.\u00a0-D., Dafflon, J., Da\u00a0Costa, P.\u00a0F., Fernandez, V., Nachev, P., Ourselin, S., & Cardoso, M.\u00a0J. (2022). Brain imaging generation with latent diffusion models. MICCAI Workshops, (pp. 117\u2013126). Springer.","DOI":"10.1007\/978-3-031-18576-2_12"},{"key":"2756_CR146","doi-asserted-by":"crossref","unstructured":"Polamreddy, L.\u00a0R., Roy, K., Yueh, S.\u00a0-H., Mahato, D., Kuppili, S., Li, J., & Zhang, Y. (2024). Leapfrog latent consistency model (llcm) for medical images generation. CoRR. arXiv:2411.15084","DOI":"10.1145\/3721201.3725428"},{"key":"2756_CR147","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., & Hochreiter, S. (2017). Gans trained by a two time-scale update rule converge to a local nash equilibrium. NeurIPS30."},{"key":"2756_CR148","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Le\u00a0Bras, R., & Choi, Y. (2021). Clipscore: A reference-free evaluation metric for image captioning. EMNLP, (pp. 7514\u20137528).","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"2756_CR149","doi-asserted-by":"crossref","unstructured":"Jayasumana, S., Ramalingam, S., Veit, A., Glasner, D., Chakrabarti, A., & Kumar, S. (2024). Rethinking fid: Towards a better evaluation metric for image generation. CVPR, (pp. 9307\u20139315).","DOI":"10.1109\/CVPR52733.2024.00889"},{"key":"2756_CR150","doi-asserted-by":"crossref","unstructured":"Lin, T.\u00a0-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C.L. (2014). Microsoft coco: Common objects in context. ECCV, (pp. 740\u2013755).","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2756_CR151","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., & Torralba, A. (2017). Scene parsing through ade20k dataset. CVPR, (pp. 633\u2013641).","DOI":"10.1109\/CVPR.2017.544"},{"key":"2756_CR152","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2016","unstructured":"Krishna, R., Zhu, Y., Groth, O., Johnson, J., Hata, K., Kravitz, J., Chen, S., Kalantidis, Y., Li, L.-J., Shamma, D. A., Bernstein, M. S., & Fei-Fei, L. (2016). Visual genome: Connecting language and vision using crowdsourced dense image annotations. International Journal of Computer Vision, 123, 32\u201373.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR153","doi-asserted-by":"crossref","unstructured":"Shao, S., Li, Z., Zhang, T., Peng, C., Yu, G., Zhang, X., Li, J., & Sun, J. (2019). Objects365: A large-scale, high-quality dataset for object detection. ICCV, (pp. 8430\u20138439).","DOI":"10.1109\/ICCV.2019.00852"},{"key":"2756_CR154","doi-asserted-by":"crossref","unstructured":"Hudson, D.\u00a0A., & Manning, C.\u00a0D. (2019). Gqa: A new dataset for real-world visual reasoning and compositional question answering. CVPR, (pp. 6700\u20136709).","DOI":"10.1109\/CVPR.2019.00686"},{"key":"2756_CR155","doi-asserted-by":"crossref","unstructured":"Kafle, K., & Kanan, C. (2017). An analysis of visual question answering algorithms. ICCV, (pp. 1965\u20131973).","DOI":"10.18653\/v1\/W17-3529"},{"issue":"7","key":"2756_CR156","doi-asserted-by":"publisher","first-page":"1956","DOI":"10.1007\/s11263-020-01316-z","volume":"128","author":"A Kuznetsova","year":"2020","unstructured":"Kuznetsova, A., Rom, H., Alldrin, N., Uijlings, J., Krasin, I., Pont-Tuset, J., Kamali, S., Popov, S., Malloci, M., Kolesnikov, A., et al. (2020). The open images dataset v4: Unified image classification, object detection, and visual relationship detection at scale. International Journal of Computer Vision, 128(7), 1956\u20131981.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR157","doi-asserted-by":"crossref","unstructured":"Wang, X., Peng, Y., Lu, L., Lu, Z., Bagheri, M., & Summers, R.\u00a0M. (2017). Chestx-ray8: Hospital-scale chest x-ray database and benchmarks on weakly-supervised classification and localization of common thorax diseases. CVPR, (pp. 2097\u20132106).","DOI":"10.1109\/CVPR.2017.369"},{"key":"2756_CR158","doi-asserted-by":"crossref","unstructured":"Liu, B., Zhan, L.\u00a0-M., Xu, L., Ma, L., Yang, Y., & Wu, X.\u00a0-M. (2021). Slake: A semantically-labeled knowledge-enhanced dataset for medical visual question answering. ISBI, (pp. 1650\u20131654). IEEE","DOI":"10.1109\/ISBI48211.2021.9434010"},{"key":"2756_CR159","unstructured":"Bai, Z., Wang, P., Xiao, T., He, T., Han, Z., Zhang, Z., & Shou, M.\u00a0Z. (2024). Hallucination of multimodal large language models: A survey. CoRR. arXiv:2404.18930"},{"key":"2756_CR160","doi-asserted-by":"crossref","unstructured":"Agrawal, H., Desai, K., Wang, Y., Chen, X., Jain, R., Johnson, M., Batra, D., Parikh, D., Lee, S., & Anderson, P. (2019). Nocaps: Novel object captioning at scale. ICCV, (pp. 8948\u20138957).","DOI":"10.1109\/ICCV.2019.00904"},{"key":"2756_CR161","first-page":"61735","volume":"37","author":"V Udandarao","year":"2024","unstructured":"Udandarao, V., Prabhu, A., Ghosh, A., Sharma, Y., Torr, P., Bibi, A., Albanie, S., & Bethge, M. (2024). No \u201czero-shot\u2019\u2019 without exponential data: Pretraining concept frequency determines multimodal model performance. NeurIPS, 37, 61735\u201361792.","journal-title":"NeurIPS"},{"key":"2756_CR162","doi-asserted-by":"crossref","unstructured":"Parashar, S., Lin, Z., Liu, T., Dong, X., Li, Y., Ramanan, D., Caverlee, J., & Kong, S. (2024). The neglected tails in vision-language models. CVPR, (pp. 12988\u201312997).","DOI":"10.1109\/CVPR52733.2024.01234"},{"key":"2756_CR163","doi-asserted-by":"crossref","unstructured":"Yu, Q., Li, J., Wei, L., Pang, L., Ye, W., Qin, B., Tang, S., Tian, Q., & Zhuang, Y. (2024). Hallucidoctor: Mitigating hallucinatory toxicity in visual instruction data. CVPR, (pp. 12944\u201312953).","DOI":"10.1109\/CVPR52733.2024.01230"},{"key":"2756_CR164","doi-asserted-by":"crossref","unstructured":"Leng, S., Zhang, H., Chen, G., Li, X., Lu, S., Miao, C., & Bing, L. (2024). Mitigating object hallucinations in large vision-language models through visual contrastive decoding. CVPR, (pp. 13872\u201313882).","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"2756_CR165","doi-asserted-by":"crossref","unstructured":"Tong, S., Liu, Z., Zhai, Y., Ma, Y., LeCun, Y., & Xie, S. (2024). Eyes wide shut? exploring the visual shortcomings of multimodal llms. CVPR, (pp. 9568\u20139578).","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"2756_CR166","doi-asserted-by":"crossref","unstructured":"Wei, H., Kong, L., Chen, J., Zhao, L., Ge, Z., Yang, J., Sun, J., Han, C., & Zhang, X. (2024). Vary: Scaling up the vision vocabulary for large vision-language model. ECCV, (pp. 408\u2013424).","DOI":"10.1007\/978-3-031-73235-5_23"},{"key":"2756_CR167","unstructured":"Li, Y., Zhou, K., Zhao, W.\u00a0X., Fang, L., & Wen, J.\u00a0-R. (2025). Analyzing and mitigating object hallucination: A training bias perspective. CoRR. arXiv:2508.04567"},{"key":"2756_CR168","unstructured":"Xie, C., Liu, T., Jiang, L., Zeng, Y., Shen, Y., Huang, W., Li, J., Xu, X., et al. (2025). Tarac: Mitigating hallucination in lvlms via temporal attention real-time accumulative connection. CoRR. arXiv:2504.04099"},{"key":"2756_CR169","unstructured":"Zhou, Y., Cui, C., Yoon, J., Zhang, L., Deng, Z., Finn, C., Bansal, M., & Yao, H. (2024). Analyzing and mitigating object hallucination in large vision-language models. ICLR."},{"key":"2756_CR170","doi-asserted-by":"crossref","unstructured":"Wu, J., Shi, Z., Wang, S., Huang, J., Yin, D., Yan, L., Cao, M., & Zhang, M. (2025). Mitigating hallucinations in large vision-language models via entity-centric multimodal preference optimization. ACL.","DOI":"10.18653\/v1\/2025.emnlp-main.982"},{"key":"2756_CR171","doi-asserted-by":"crossref","unstructured":"Liu, Q., Shang, C., Liu, L., Pappas, N., Ma, J., John, N.\u00a0A., Doss, S., Marquez, L., Ballesteros, M., & Benajiba, Y. (2025). Unraveling and mitigating safety alignment degradation of vision-language models. ACL, (pp. 3631\u20133643).","DOI":"10.18653\/v1\/2025.findings-acl.186"},{"key":"2756_CR172","doi-asserted-by":"crossref","unstructured":"Alonso, I., Azkune, G., Salaberria, A., Barnes, J., & Lacalle, O.L. (2025). Vision-language models struggle to align entities across modalities. CoRR. arXiv:2503.03854","DOI":"10.18653\/v1\/2025.findings-acl.965"},{"key":"2756_CR173","unstructured":"Zheng, H., & Zhang, Z. (2025). Modality bias in lvlms: Analyzing and mitigating object hallucination via attention lens. CoRR. arXiv:2508.02419"},{"key":"2756_CR174","doi-asserted-by":"crossref","unstructured":"Xie, Y., Li, G., Xu, X., & Kan, M.\u00a0-Y. (2024). V-dpo: Mitigating hallucination in large vision language models via vision-guided direct preference optimization. EMNLP, (pp. 13258\u201313273).","DOI":"10.18653\/v1\/2024.findings-emnlp.775"},{"key":"2756_CR175","doi-asserted-by":"crossref","unstructured":"Li, S., Qu, J., Zhou, Y., Qin, Y., Yang, T., & Zhao, Y. (2025). Treble counterfactual vlms: A causal approach to hallucination. CoRR. arXiv:2503.06169","DOI":"10.18653\/v1\/2025.findings-emnlp.1000"},{"key":"2756_CR176","doi-asserted-by":"crossref","unstructured":"Abaid, A., Farooq, M.\u00a0A., Hynes, N., Corcoran, P., & Ullah, I. (2024). Synthesizing cta image data for type-b aortic dissection using stable diffusion models. IEEE EMBC, (pp. 1\u20135).","DOI":"10.1109\/EMBC53108.2024.10782969"},{"key":"2756_CR177","doi-asserted-by":"crossref","unstructured":"Bui, N.\u00a0-T., Hoang, D.\u00a0-H., Trinh, Q.\u00a0-H., Tran, M.\u00a0-T., Nguyen, T., & Gauch, S. (2025). Nein: Telling what you don\u2019t want. CVPR, (pp. 3107\u20133115).","DOI":"10.1109\/CVPRW67362.2025.00293"},{"key":"2756_CR178","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., & Efros, A.\u00a0A. (2023). Instructpix2pix: Learning to follow image editing instructions. CVPR, (pp. 18392\u201318402).","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"2756_CR179","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., & Soricut, R. (2018). Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. ACL, (pp. 2556\u20132565).","DOI":"10.18653\/v1\/P18-1238"},{"key":"2756_CR180","unstructured":"Conwell, C., Tawiah-Quashie, R., & Ullman, T. (2024). Relations, negations, and numbers: Looking for logic in generative text-to-image models. CoRR. arXiv:2411.17066"},{"key":"2756_CR181","doi-asserted-by":"crossref","unstructured":"Vice, J., Akhtar, N., Hartley, R., & Mian, A. (2025). Exploring bias in over 100 text-to-image generative models. CoRR. arXiv:2503.08012","DOI":"10.1109\/TDSC.2025.3572115"},{"key":"2756_CR182","first-page":"70376","volume":"36","author":"Z Huang","year":"2023","unstructured":"Huang, Z., Zhou, P., Yan, S., & Lin, L. (2023). Scalelong: Towards more stable training of diffusion model via scaling network long skip connection. NeurIPS, 36, 70376\u201370401.","journal-title":"NeurIPS"},{"issue":"4","key":"2756_CR183","first-page":"1","volume":"42","author":"H Chefer","year":"2023","unstructured":"Chefer, H., Alaluf, Y., Vinker, Y., Wolf, L., & Cohen-Or, D. (2023). Attend-and-excite: Attention-based semantic guidance for text-to-image diffusion models. Association for Computing Machinery Transactions on Graphics, 42(4), 1\u201310.","journal-title":"Association for Computing Machinery Transactions on Graphics"},{"key":"2756_CR184","doi-asserted-by":"crossref","unstructured":"Phung, Q., Ge, S., & Huang, J.\u00a0-B. (2024). Grounded text-to-image synthesis with attention refocusing. CVPR, (pp. 7932\u20137942).","DOI":"10.1109\/CVPR52733.2024.00758"},{"key":"2756_CR185","unstructured":"Jiang, Y., Gupta, A., Zhang, Z., Wang, G., Dou, Y., Chen, Y., Fei-Fei, L., Anandkumar, A., Zhu, Y., & Fan, L. (2022). Vima: General robot manipulation with multimodal prompts. NeurIPS Workshops."},{"key":"2756_CR186","first-page":"32340","volume":"35","author":"A Majumdar","year":"2022","unstructured":"Majumdar, A., Aggarwal, G., Devnani, B., Hoffman, J., & Batra, D. (2022). Zson: Zero-shot object-goal navigation using multimodal goal embeddings. NeurIPS, 35, 32340\u201332352.","journal-title":"NeurIPS"},{"key":"2756_CR187","doi-asserted-by":"crossref","unstructured":"Guo, Z., Yagudin, Z., Lykov, A., Konenkov, M., & Tsetserukou, D. (2024). Vlm-auto: Vlm-based autonomous driving assistant with human-like behavior and understanding for complex road scenes. FLLM, (pp. 501\u2013507). IEEE","DOI":"10.1109\/FLLM63129.2024.10852498"},{"key":"2756_CR188","unstructured":"Xu, M., Huang, P., Yu, W., Liu, S., Zhang, X., Niu, Y., Zhang, T., Xia, F., Tan, J., & Zhao, D. (2023). Creative robot tool use with large language models. CoRR. arXiv:2310.13065"},{"key":"2756_CR189","unstructured":"Shah, R., Mart\u00edn-Mart\u00edn, R., & Zhu, Y. (2023). Mutex: Learning unified policies from multimodal task specifications. CoRL."},{"key":"2756_CR190","unstructured":"Yuan, Z., Zhang, T., Deng, Y., Zhang, J., Zhu, Y., Jia, Z., Zhou, J., & Zhang, J. (2024). Walkvlm: Aid visually impaired people walking by vision language model. CoRR. arXiv:2412.20903"},{"key":"2756_CR191","unstructured":"Xiao, J., Huang, N., Qiu, H., Tao, Z., Yang, X., Hong, R., Wang, M., & Yao, A. (2025). Egoblind: Towards egocentric visual assistance for the blind people. CoRR. arXiv:2503.08221"},{"issue":"5","key":"2756_CR192","first-page":"4542","volume":"38","author":"T Qian","year":"2024","unstructured":"Qian, T., Chen, J., Zhuo, L., Jiao, Y., & Jiang, Y.-G. (2024). Nuscenes-qa: A multi-modal visual question answering benchmark for autonomous driving scenario. Association for the Advancement of Artificial Intelligence, 38(5), 4542\u20134550.","journal-title":"Association for the Advancement of Artificial Intelligence"},{"key":"2756_CR193","doi-asserted-by":"crossref","unstructured":"Yin, T., Zhou, X., & Krahenbuhl, P. (2021). Center-based 3d object detection and tracking. CVPR, (pp. 11784\u201311793).","DOI":"10.1109\/CVPR46437.2021.01161"},{"key":"2756_CR194","doi-asserted-by":"crossref","unstructured":"Jiao, Y., Jie, Z., Chen, S., Chen, J., Ma, L., & Jiang, Y.\u00a0-G. (2023). Msmdfusion: Fusing lidar and camera at multiple scales with multi-depth seeds for 3d object detection. CVPR, (pp. 21643\u201321652).","DOI":"10.1109\/CVPR52729.2023.02073"},{"key":"2756_CR195","unstructured":"Long, K., Guo, J., Zhang, T., Yu, H., & Li, X. (2025). A low-rank method for vision language model hallucination mitigation in autonomous driving. CoRR. arXiv:2511.06496"},{"key":"2756_CR196","doi-asserted-by":"crossref","unstructured":"Shin, S., Jeon, S., Kim, J., Kang, G.\u00a0-C., & Zhang, B.\u00a0-T. (2024). Socratic planner: Inquiry-based zero-shot planning for embodied instruction following. CoRR.","DOI":"10.1109\/ICRA55743.2025.11128677"},{"key":"2756_CR197","unstructured":"Wang, Z., Liu, A., Lin, H., Li, J., Ma, X., & Liang, Y. (2024). Rat: Retrieval augmented thoughts elicit context-aware reasoning in long-horizon generation. CoRR. arXiv:2403.05313"},{"key":"2756_CR198","unstructured":"Moor, M., Huang, Q., Wu, S., Yasunaga, M., Dalmia, Y., Leskovec, J., Zakka, C., Reis, E.\u00a0P., & Rajpurkar, P. (2023). Med-flamingo: A multimodal medical few-shot learner. ML4H, (pp. 353\u2013367)."},{"issue":"1","key":"2756_CR199","doi-asserted-by":"publisher","first-page":"7866","DOI":"10.1038\/s41467-025-62385-7","volume":"16","author":"C Wu","year":"2025","unstructured":"Wu, C., Zhang, X., Zhang, Y., Hui, H., Wang, Y., & Xie, W. (2025). Towards generalist foundation model for radiology by leveraging web-scale 2d &3d medical data. Nature Communications, 16(1), 7866.","journal-title":"Nature Communications"},{"key":"2756_CR200","doi-asserted-by":"crossref","unstructured":"Chen, J., Gui, C., Ouyang, R., Gao, A., Chen, S., Chen, G.\u00a0H., Wang, X., Cai, Z., Ji, K., Wan, X., et al. (2024). Towards injecting medical visual knowledge into multimodal llms at scale. EMNLP, (pp. 7346\u20137370).","DOI":"10.18653\/v1\/2024.emnlp-main.418"},{"key":"2756_CR201","first-page":"28541","volume":"36","author":"C Li","year":"2023","unstructured":"Li, C., Wong, C., Zhang, S., Usuyama, N., Liu, H., Yang, J., Naumann, T., Poon, H., & Gao, J. (2023). Llava-med: Training a large language-and-vision assistant for biomedicine in one day. NeurIPS, 36, 28541\u201328564.","journal-title":"NeurIPS"},{"key":"2756_CR202","unstructured":"Yan, Q., Yuan, Y., Hu, X., Wang, Y., Xu, J., Li, J., Fu, C.\u00a0-W., & Heng, P.\u00a0-A. (2025). Medhalltune: An instruction-tuning benchmark for mitigating medical hallucination in vision-language models. CoRR. arXiv:2502.20780"},{"issue":"1","key":"2756_CR203","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/sdata.2018.251","volume":"5","author":"JJ Lau","year":"2018","unstructured":"Lau, J. J., Gayen, S., Ben Abacha, A., & Demner-Fushman, D. (2018). A dataset of clinically generated visual questions and answers about radiology images. Scientific Data, 5(1), 1\u201310.","journal-title":"Scientific Data"},{"key":"2756_CR204","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2020.3042202","volume":"60","author":"R Zhao","year":"2021","unstructured":"Zhao, R., Shi, Z., & Zou, Z. (2021). High-resolution remote sensing image captioning based on structured attention. IEEE Transactions on Geoscience and Remote Sensing, 60, 1\u201314.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"2756_CR205","first-page":"1","volume":"62","author":"F Liu","year":"2024","unstructured":"Liu, F., Chen, D., Guan, Z., Zhou, X., Zhu, J., Ye, Q., Fu, L., & Zhou, J. (2024). Remoteclip: A vision language foundation model for remote sensing. IEEE Transactions on Geoscience and Remote Sensing, 62, 1\u201316.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"issue":"6","key":"2756_CR206","first-page":"5481","volume":"38","author":"J Wang","year":"2024","unstructured":"Wang, J., Zheng, Z., Chen, Z., Ma, A., & Zhong, Y. (2024). Earthvqa: Towards queryable earth via relational reasoning-based remote sensing visual question answering. Association for the Advancement of Artificial Intelligence, 38(6), 5481\u20135489.","journal-title":"Association for the Advancement of Artificial Intelligence"},{"key":"2756_CR207","doi-asserted-by":"publisher","first-page":"272","DOI":"10.1016\/j.isprsjprs.2025.03.028","volume":"224","author":"Y Hu","year":"2025","unstructured":"Hu, Y., Yuan, J., Wen, C., Lu, X., Liu, Y., & Li, X. (2025). Rsgpt: A remote sensing vision language model and benchmark. ISPRS Journal of Photogrammetry and Remote Sensing, 224, 272\u2013286.","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"2756_CR208","first-page":"8815","volume":"36","author":"D Wang","year":"2023","unstructured":"Wang, D., Zhang, J., Du, B., Xu, M., Liu, L., Tao, D., & Zhang, L. (2023). Samrs: Scaling-up remote sensing segmentation dataset with segment anything model. NeurIPS, 36, 8815\u20138827.","journal-title":"NeurIPS"},{"key":"2756_CR209","doi-asserted-by":"crossref","unstructured":"Du, D., Zhu, P., Wen, L., Bian, X., Lin, H., Hu, Q., Peng, T., Zheng, J., Wang, X., Zhang, Y., et al. (2019). Visdrone-det2019: The vision meets drone object detection in image challenge results. ICCV Workshops.","DOI":"10.1109\/ICCVW.2019.00030"},{"issue":"4","key":"2756_CR210","doi-asserted-by":"publisher","first-page":"719","DOI":"10.3390\/rs17040719","volume":"17","author":"H Li","year":"2025","unstructured":"Li, H., Zhang, X., & Qu, H. (2025). Ddfav: Remote sensing large vision language models dataset and evaluation benchmark. Remote Sensing, 17(4), 719.","journal-title":"Remote Sensing"},{"issue":"10","key":"2756_CR211","doi-asserted-by":"publisher","first-page":"7128","DOI":"10.1007\/s11263-025-02521-4","volume":"133","author":"T-D Truong","year":"2025","unstructured":"Truong, T.-D., Nguyen, H.-Q., Nguyen, X.-B., Dowling, A., Li, X., & Luu, K. (2025). Insect-foundation: A foundation model and large multimodal dataset for vision-language insect understanding. International Journal of Computer Vision, 133(10), 7128\u20137153.","journal-title":"International Journal of Computer Vision"},{"key":"2756_CR212","unstructured":"Li, L., Li, J., Chen, D., Pu, L., Yao, H., & Huang, Y. (2025). Vllfl: A vision-language model based lightweight federated learning framework for smart agriculture. CoRR. arXiv:2504.13365"},{"issue":"08","key":"2756_CR213","doi-asserted-by":"publisher","first-page":"5227","DOI":"10.1109\/TPAMI.2024.3362475","volume":"46","author":"D Hong","year":"2024","unstructured":"Hong, D., Zhang, B., Li, X., Li, Y., Li, C., Yao, J., Yokoya, N., Li, H., Ghamisi, P., Jia, X., et al. (2024). Spectralgpt: Spectral remote sensing foundation model. IEEE Transactions on Pattern Analysis and Machine Intelligence, 46(08), 5227\u20135244.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2756_CR214","unstructured":"Zhao, X., Lyu, X., & Li, X. (2025). Agrivln: Vision-and-language navigation for agricultural robots. CoRR. arXiv:2508.07406"},{"issue":"7","key":"2756_CR215","first-page":"7641","volume":"38","author":"G Zhou","year":"2024","unstructured":"Zhou, G., Hong, Y., & Wu, Q. (2024). Navgpt: Explicit reasoning in vision-and-language navigation with large language models. Association for the Advancement of Artificial Intelligence, 38(7), 7641\u20137649.","journal-title":"Association for the Advancement of Artificial Intelligence"},{"key":"2756_CR216","unstructured":"Chawla, S., Singh, N., & Drori, I. (2021). Quantifying and alleviating distribution shifts in foundation models on review classification. NeurIPS Workshops."},{"key":"2756_CR217","doi-asserted-by":"crossref","unstructured":"Ojha, J., Presacan, O., G.\u00a0Lind, P., Monteiro, E., & Yazidi, A.: Navigating uncertainty: A user-perspective survey of trustworthiness of ai in healthcare. ACM Transactions on Computing for Healthcare 6(3), 1\u201332 (2025)","DOI":"10.1145\/3716317"},{"key":"2756_CR218","unstructured":"Noorani, S., Kiyani, S., Pappas, G., & Hassani, H. (2025). Human-ai collaborative uncertainty quantification. CoRR arXiv:2510.23476"},{"key":"2756_CR219","doi-asserted-by":"crossref","unstructured":"Chakraborty, N., Ornik, M., & Driggs-Campbell, K. (2025). Hallucination detection in foundation models for decision-making: A flexible definition and review of the state of the art. Surv: ACM Comput.","DOI":"10.1145\/3716846"},{"key":"2756_CR220","unstructured":"Yan, B., Zhang, J., Yuan, Z., Shan, S., & Chen, X. (2024). Evaluating the quality of hallucination benchmarks for large vision-language models. CoRR. arXiv:2406.17115"},{"key":"2756_CR221","doi-asserted-by":"crossref","unstructured":"Li, Q., Geng, J., Lyu, C., Zhu, D., Panov, M., & Karray, F. (2024). Reference-free hallucination detection for large vision-language models. EMNLP, (pp. 4542\u20134551).","DOI":"10.18653\/v1\/2024.findings-emnlp.262"},{"key":"2756_CR222","unstructured":"Song, Y., Qiu, L., Zhang, X., & Tang, Z. (2025). Hallucination detection via internal states and structured reasoning consistency in large language models. arXiv:2510.11529"},{"key":"2756_CR223","doi-asserted-by":"publisher","first-page":"1220476","DOI":"10.3389\/frai.2023.1220476","volume":"6","author":"M Cafagna","year":"2023","unstructured":"Cafagna, M., Rojas-Barahona, L. M., Deemter, K., & Gatt, A. (2023). Interpreting vision and language generative models with semantic visual priors. Frontiers in Artificial Intelligence, 6, 1220476.","journal-title":"Frontiers in Artificial Intelligence"},{"key":"2756_CR224","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2023.104771","volume":"137","author":"A Borji","year":"2023","unstructured":"Borji, A. (2023). Qualitative failures of image generation models and their application in detecting deepfakes. Image and Vision Computing, 137, Article 104771.","journal-title":"Image and Vision Computing"},{"key":"2756_CR225","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., & Cohen-Or, D. (2023). Prompt-to-prompt image editing with cross-attention control. ICLR."},{"key":"2756_CR226","first-page":"25081","volume":"36","author":"Y Mu","year":"2023","unstructured":"Mu, Y., Zhang, Q., Hu, M., Wang, W., Ding, M., Jin, J., Wang, B., Dai, J., Qiao, Y., & Luo, P. (2023). Embodiedgpt: Vision-language pre-training via embodied chain of thought. NeurIPS, 36, 25081\u201325094.","journal-title":"NeurIPS"},{"key":"2756_CR227","unstructured":"Shao, H., Qian, S., Xiao, H., Song, G., Zong, Z., Wang, L., Liu, Y., & Li, H. (2024). Visual cot: Advancing multi-modal language models with a comprehensive dataset and benchmark for chain-of-thought reasoning. NeurIPS."},{"key":"2756_CR228","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Lu, Y., Kim, M.\u00a0J., Fu, Z., Zhang, Z., Wu, Y., Li, Z., Ma, Q., Han, S., Finn, C., et al. (2025). Cot-vla: Visual chain-of-thought reasoning for vision-language-action models. CVPR, (pp. 1702\u20131713).","DOI":"10.1109\/CVPR52734.2025.00166"},{"key":"2756_CR229","doi-asserted-by":"crossref","unstructured":"Rawte, V., Mishra, A., Sheth, A., & Das, A. (2025). Defining and quantifying visual hallucinations in vision-language models. TrustNLP Workshop, (pp. 501\u2013510).","DOI":"10.18653\/v1\/2025.trustnlp-main.32"},{"key":"2756_CR230","first-page":"51503","volume":"37","author":"Y Zhou","year":"2024","unstructured":"Zhou, Y., Fan, Z., Cheng, D., Yang, S., Chen, Z., Cui, C., Wang, X., Li, Y., Zhang, L., & Yao, H. (2024). Calibrated self-rewarding vision language models. NeurIPS, 37, 51503\u201351531.","journal-title":"NeurIPS"},{"key":"2756_CR231","doi-asserted-by":"crossref","unstructured":"Kang, G., Gao, W., & Zhan, J. (2025). Evaluatology-driven artificial intelligence. TBench, (pp. 100245).","DOI":"10.1016\/j.tbench.2025.100245"},{"issue":"5","key":"2756_CR232","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1038\/s42256-019-0048-x","volume":"1","author":"C Rudin","year":"2019","unstructured":"Rudin, C. (2019). Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead. Nature Machine Intelligence, 1(5), 206\u2013215.","journal-title":"Nature Machine Intelligence"},{"key":"2756_CR233","doi-asserted-by":"crossref","unstructured":"Yang, Y., Panagopoulou, A., Zhou, S., Jin, D., Callison-Burch, C., & Yatskar, M. (2023). Language in a bottle: Language model guided concept bottlenecks for interpretable image classification. CVPR, (pp. 19187\u201319197).","DOI":"10.1109\/CVPR52729.2023.01839"},{"key":"2756_CR234","unstructured":"Ben Melech\u00a0Stan, G., Aflalo, E., Rohekar, R.\u00a0Y., Bhiwandiwalla, A., Tseng, S.\u00a0-Y., Olson, M.\u00a0L., Gurwicz, Y., Wu, C., Duan, N., & Lal, V. (2024). Lvlm-intrepret: An interpretability tool for large vision-language models. CVPR, (pp. 8182\u20138187)."},{"key":"2756_CR235","doi-asserted-by":"crossref","unstructured":"Wu, K., Xu, S., Chen, H., Wang, C., Li, Z., Wang, Y., & Zhong, F. (2025). Vlm can be a good assistant: Enhancing embodied visual tracking with self-improving visual-language models. CoRR. arXiv:2505.20718","DOI":"10.1109\/IROS60139.2025.11246600"},{"key":"2756_CR236","first-page":"75942","volume":"37","author":"G Sarch","year":"2024","unstructured":"Sarch, G., Jang, L., Tarr, M., Cohen, W. W., Marino, K., & Fragkiadaki, K. (2024). Vlm agents generate their own memories: Distilling experience into embodied programs of thought. NeurIPS, 37, 75942\u201375985.","journal-title":"NeurIPS"},{"key":"2756_CR237","unstructured":"Kalai, A.\u00a0T., Nachum, O., Vempala, S.\u00a0S., & Zhang, E. (2025). Why language models hallucinate. CoRR. arXiv:2509.04664"},{"key":"2756_CR238","doi-asserted-by":"crossref","unstructured":"Yu, X., Cheng, H., Liu, X., Roth, D., & Gao, J. (2024). Reeval: Automatic hallucination evaluation for retrieval-augmented large language models via transferable adversarial attacks. NAACL, (pp. 1333\u20131351).","DOI":"10.18653\/v1\/2024.findings-naacl.85"},{"key":"2756_CR239","unstructured":"Zhu, Z., Yang, Y., & Sun, Z. (2024). Halueval-wild: Evaluating hallucinations of language models in the wild. CoRR. arXiv:2403.04307"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02756-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02756-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02756-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T10:07:21Z","timestamp":1774606041000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02756-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":239,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["2756"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02756-9","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"25 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"131"}}