{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T08:10:26Z","timestamp":1783930226895,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":46,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234318","type":"print"},{"value":"9789819234325","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3432-5_48","type":"book-chapter","created":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T07:41:07Z","timestamp":1783928467000},"page":"588-597","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Look Before You Speak: Visual Re-Focusing for Hallucination Mitigation in Large Vision-Language Models"],"prefix":"10.1007","author":[{"given":"Zheyuan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingmin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Shu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"key":"48_CR1","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, PMLR, pp. 19730\u201319742 (2023)"},{"key":"48_CR2","unstructured":"Bai, J., et al.: Qwen-VL: a frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)"},{"key":"48_CR3","doi-asserted-by":"crossref","unstructured":"Ye, Q., et al.: mPLUG-Owl2: revolutionizing multi-modal large language model with modality collaboration. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 13040\u201313051 (2024)","DOI":"10.1109\/CVPR52733.2024.01239"},{"key":"48_CR4","doi-asserted-by":"publisher","unstructured":"Wu, Y., et al.: DetToolChain: a new prompting paradigm to\u00a0unleash detection ability of\u00a0MLLM. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) ECCV 2024. LNCS, vol. 15090, pp. 164\u2013182. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-73411-3_10","DOI":"10.1007\/978-3-031-73411-3_10"},{"key":"48_CR5","doi-asserted-by":"crossref","unstructured":"Zang, Y., Li, W., Han, J., Zhou, K., Loy, C.C.: Contextual object detection with multimodal large language models. Int. J. Comput. Vis., 1\u201319 (2024)","DOI":"10.1007\/s11263-024-02214-4"},{"key":"48_CR6","doi-asserted-by":"crossref","unstructured":"Wan, Z., et al.: InstructPart: task-oriented part segmentation with instruction reasoning. arXiv preprint arXiv:2505.18291 (2025)","DOI":"10.18653\/v1\/2025.acl-long.1179"},{"key":"48_CR7","doi-asserted-by":"crossref","unstructured":"Lai, X., et al.: LISA: reasoning segmentation via large language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9579\u20139589 (2024)","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"48_CR8","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"48_CR9","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Hendricks, L.A., Burns, K., Darrell, T., Saenko, K.: Object hallucination in image captioning. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 4035\u20134045 (2018)","DOI":"10.18653\/v1\/D18-1437"},{"key":"48_CR10","unstructured":"Liu, H., et al.: A survey on hallucination in large vision-language models. arXiv preprint arXiv:2402.00253 (2024)"},{"key":"48_CR11","unstructured":"Bai, Z., et al.: Hallucination of multimodal large language models: a survey. arXiv preprint arXiv:2404.18930 (2024)"},{"key":"48_CR12","unstructured":"Chen, J., et al.: Detecting and evaluating medical hallucinations in large vision language models. arXiv preprint arXiv:2406.10185 (2024)"},{"key":"48_CR13","doi-asserted-by":"publisher","unstructured":"Zhang, J., Wang, T., Zhang, H., Lu, P., Zheng, F.: Reflective instruction tuning: mitigating hallucinations in\u00a0large vision-language models. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) ECCV 2024. LNCS, vol. 15126, pp. 196\u2013213. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-73113-6_12. https:\/\/www.ecva.net\/papers\/eccv_2024\/papers_ECCV\/papers\/08550.pdf","DOI":"10.1007\/978-3-031-73113-6_12"},{"key":"48_CR14","unstructured":"Chen, Z., et al.: Mitigating hallucination in visual language models with visual supervision. arXiv preprint arXiv:2311.16479 (2023)"},{"key":"48_CR15","doi-asserted-by":"crossref","unstructured":"Sun, Z., et al.: Aligning large multimodal models with factually augmented RLHF. arXiv preprint arXiv:2309.14525 (2023)","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"48_CR16","doi-asserted-by":"crossref","unstructured":"Li, X.L., et al.: Contrastive decoding: open-ended text generation as optimization. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 12286\u201312312 (2023)","DOI":"10.18653\/v1\/2023.acl-long.687"},{"key":"48_CR17","unstructured":"Chuang, Y.-S., Xie, Y., Luo, H., Kim, Y., Glass, J.R., He, P.: DoLa: decoding by contrasting layers improves factuality in large language models. In: International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=Th6NyL07na"},{"key":"48_CR18","doi-asserted-by":"crossref","unstructured":"Wang, X., Pan, J., Ding, L., Biemann, C.: Mitigating hallucinations in large vision-language models with instruction contrastive decoding. In: Findings of the Association for Computational Linguistics ACL 2024, pp. 15840\u201315853 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.937"},{"key":"48_CR19","doi-asserted-by":"crossref","unstructured":"Favero, A., et al.: Multi-modal hallucination control by visual information grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14303\u201314312 (2024)","DOI":"10.1109\/CVPR52733.2024.01356"},{"key":"48_CR20","doi-asserted-by":"crossref","unstructured":"Leng, S., et al.: Mitigating object hallucinations in large vision-language models through visual contrastive decoding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13872\u201313882 (2024)","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"48_CR21","doi-asserted-by":"publisher","unstructured":"Chen, L., et al.: An image is worth 1\/2 tokens after layer 2: plug-and-play inference acceleration for\u00a0large vision-language models. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) ECCV 2024. LNCS, vol. 15139, pp. 19\u201335. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-73004-7_2","DOI":"10.1007\/978-3-031-73004-7_2"},{"key":"48_CR22","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26296\u201326306 (2024)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"48_CR23","doi-asserted-by":"publisher","first-page":"49250","DOI":"10.52202\/075280-2142","volume":"36","author":"W Dai","year":"2023","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 49250\u201349267 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"48_CR24","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Evaluating object hallucination in large vision-language models. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 292\u2013305 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"48_CR25","unstructured":"Fu, C., et al.: MME: a comprehensive evaluation benchmark for multimodal large language models. arXiv preprint arXiv:2306.13394 (2023)"},{"key":"48_CR26","doi-asserted-by":"publisher","unstructured":"Liu, Y., et al.: MMBench: is your multi-modal model an\u00a0all-around player?. In: Leonardis, A., Ricci, E., Roth, S., Russakovsky, O., Sattler, T., Varol, G. (eds.) ECCV 2024. LNCS, vol. 15064, pp. 216\u2013233. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-72658-3_13","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"48_CR27","unstructured":"Yu, W., et al.: MM-Vet: evaluating large multimodal models for integrated capabilities. In: International Conference on Machine Learning (2024). https:\/\/openreview.net\/forum?id=KOTutrSR2y"},{"key":"48_CR28","doi-asserted-by":"crossref","unstructured":"Tong, S., Liu, Z., Zhai, Y., Ma, Y., LeCun, Y., Xie, S.: Eyes wide shut? Exploring the visual shortcomings of multimodal LLMs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9568\u20139578 (2024)","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"48_CR29","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"48_CR30","unstructured":"Chiang, W.-L., et al.: Vicuna: an open-source chatbot impressing GPT-4 with 90%ChatGPT quality (2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"48_CR31","doi-asserted-by":"publisher","first-page":"34892","DOI":"10.52202\/075280-1516","volume":"36","author":"H Liu","year":"2023","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 34892\u201334916 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"48_CR32","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, PMLR, pp. 8748\u20138763 (2021)"},{"key":"48_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, C., et al.: Self-correcting decoding with generative feedback for mitigating hallucinations in large vision-language models. In: International Conference on Learning Representations (2025). https:\/\/openreview.net\/forum?id=tTBXePRKSx","DOI":"10.1109\/EIECS67708.2025.11283452"},{"key":"48_CR34","unstructured":"Zhang, C., et al.: Incorporating generative feedback for mitigating hallucinations in large vision-language models. In: Workshop on Responsibly Building the Next Generation of Multimodal Foundational Models (2024)"},{"key":"48_CR35","doi-asserted-by":"publisher","first-page":"122811","DOI":"10.52202\/079017-3902","volume":"37","author":"X Lyu","year":"2024","unstructured":"Lyu, X., Chen, B., Gao, L., Shen, H., Song, J.: Alleviating hallucinations in large vision-language models through hallucination-induced optimization. Adv. Neural. Inf. Process. Syst. 37, 122811\u2013122832 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"48_CR36","doi-asserted-by":"crossref","unstructured":"Jiang, C., et al.: Hallucination augmented contrastive learning for multimodal large language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27036\u201327046 (2024)","DOI":"10.1109\/CVPR52733.2024.02553"},{"key":"48_CR37","unstructured":"Chen, Z., Zhao, Z., Luo, H., Yao, H., Li, B., Zhou, J.: HALC: object hallucination reduction via adaptive focal-contrast decoding. In: International Conference on Machine Learning, PMLR, pp. 7824\u20137846 (2024)"},{"key":"48_CR38","unstructured":"Liu, S., Ye, H., Xing, L.: Reducing hallucinations in vision-language models via latent space steering (2024)"},{"key":"48_CR39","unstructured":"Yin, J., Chen, Q., Chen, K.: Dynamic multimodal activation steering for hallucination mitigation in large vision-language models (2026)"},{"key":"48_CR40","doi-asserted-by":"crossref","unstructured":"An, W., Tian, F., Leng, S.: Mitigating object hallucinations in large vision-language models with assembly of global and local attention (2025)","DOI":"10.1109\/CVPR52734.2025.02784"},{"key":"48_CR41","unstructured":"Li, Q., Ye, Z., Feng, X.: CAI: caption-sensitive attention intervention for mitigating object hallucination in large vision-language models (2025)"},{"key":"48_CR42","unstructured":"Fazli, M., Wei, B., Zhu, Z.: Mitigating hallucination in large vision-language models via adaptive attention calibration (2025)"},{"key":"48_CR43","unstructured":"Bu, W., Yuan, G., Zhang, G.: Conscious gaze: adaptive attention mechanisms for hallucination mitigation in vision-language models (2025)"},{"key":"48_CR44","doi-asserted-by":"crossref","unstructured":"Li, W., Huang, Z., Li, H.: Visual evidence prompting mitigates hallucinations in large vision-language models (2025)","DOI":"10.18653\/v1\/2025.acl-long.205"},{"key":"48_CR45","unstructured":"Yin, S., et al.: Woodpecker: hallucination correction for multimodal large language models. arXiv preprint arXiv:2310.16045 (2023)"},{"key":"48_CR46","doi-asserted-by":"crossref","unstructured":"Huang, Q., et al.: Opera: alleviating hallucination in multi-modal large language models via over-trust penalty and retrospection-allocation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13418\u201313427 (2024)","DOI":"10.1109\/CVPR52733.2024.01274"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3432-5_48","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T07:41:14Z","timestamp":1783928474000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3432-5_48"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,14]]},"ISBN":["9789819234318","9789819234325"],"references-count":46,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3432-5_48","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,14]]},"assertion":[{"value":"14 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}