{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T20:04:16Z","timestamp":1784145856068,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":24,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819235063","type":"print"},{"value":"9789819235070","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T00:00:00Z","timestamp":1784160000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T00:00:00Z","timestamp":1784160000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3507-0_41","type":"book-chapter","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T19:35:10Z","timestamp":1784144110000},"page":"493-504","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Guided by LLM, Grounded in Targeted Vision: Dynamic Chain-of-Thought with Retrieval-Augmented In-Context Learning for Knowledge-Based Visual Question Answering"],"prefix":"10.1007","author":[{"given":"Yuling","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weizhuo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fangfang","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuai","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanbing","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,16]]},"reference":[{"key":"41_CR1","first-page":"2425","volume-title":"Proceedings of the IEEE International Conference on Computer Vision","author":"S Antol","year":"2015","unstructured":"Antol, S., et al.: VQA: Visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)"},{"key":"41_CR2","first-page":"256","volume-title":"European Conference on Computer Vision","author":"C Sima","year":"2024","unstructured":"Sima, C., et al.: Drivelm: driving with graph visual question answering. In: European Conference on Computer Vision, pp. 256\u2013274. Springer (2024)"},{"key":"41_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102270","volume":"106","author":"MF Ishmam","year":"2024","unstructured":"Ishmam, M.F., Shovon, M.S.H., Mridha, M.F., Dey, N.: From image to language: a critical analysis of visual question answering (VQA) approaches, challenges, and opportunities. Inf. Fusion. 106, 102270 (2024)","journal-title":"Inf. Fusion"},{"key":"41_CR4","first-page":"2712","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"J Wu","year":"2022","unstructured":"Wu, J., Lu, J., Sabharwal, A., Mottaghi, R.: Multi-modal answer validation for knowledge-based VQA. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 2712\u20132721 (2022)"},{"key":"41_CR5","first-page":"19730","volume-title":"International Conference on Machine Learning","author":"J Li","year":"2023","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pretraining with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"41_CR6","first-page":"10867","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Guo","year":"2023","unstructured":"Guo, J., et al.: From images to textual prompts: zero-shot visual question answering with frozen large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10867\u201310877 (2023)"},{"key":"41_CR7","first-page":"3081","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Z Yang","year":"2022","unstructured":"Yang, Z., et al.: An empirical study of GPT-3 for few-shot knowledge-based VQA. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 3081\u20133089 (2022)"},{"key":"41_CR8","first-page":"14974","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Shao","year":"2023","unstructured":"Shao, Z., Yu, Z., Wang, M., Yu, J.: Prompting large language models with answer heuristics for knowledge-based visual question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14974\u201314983 (2023)"},{"issue":"8","key":"41_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3711680","volume":"57","author":"J Kuang","year":"2025","unstructured":"Kuang, J., et al.: Natural language understanding and inference with MLLM in visual question answering: a survey. ACM Comput. Surv. 57(8), 1\u201336 (2025)","journal-title":"ACM Comput. Surv."},{"key":"41_CR10","first-page":"146","volume-title":"European Conference on Computer Vision","author":"D Schwenk","year":"2022","unstructured":"Schwenk, D., Khandelwal, A., Clark, C., Marino, K., Mottaghi, R.: A-OKVQA: a benchmark for visual question answering using world knowledge. In: European Conference on Computer Vision, pp. 146\u2013162. Springer (2022)"},{"key":"41_CR11","doi-asserted-by":"publisher","first-page":"2061","DOI":"10.1145\/3503161.3547870","volume-title":"Proceedings of the 30th ACM International Conference on Multimedia","author":"Y Guo","year":"2022","unstructured":"Guo, Y., et al.: A unified end-to-end retriever-reader framework for knowledge-based VQA. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 2061\u20132069 (2022)"},{"key":"41_CR12","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1109\/ICME55011.2023.00011","volume-title":"2023 IEEE International Conference on Multimedia and Expo (ICME)","author":"J You","year":"2023","unstructured":"You, J., Yang, Z., Li, Q., Liu, W.: A retriever-reader framework with visual entity linking for knowledge-based visual question answering. In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 13\u201318. IEEE (2023)"},{"key":"41_CR13","doi-asserted-by":"publisher","first-page":"10939","DOI":"10.18653\/v1\/2024.emnlp-main.613","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","author":"P Jian","year":"2024","unstructured":"Jian, P., Yu, D., Zhang, J.: Large language models know what is key visual entity: an LLM-assisted multimodal retrieval for VQA. In: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 10939\u201310956 (2024)"},{"key":"41_CR14","first-page":"2963","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Y Hu","year":"2023","unstructured":"Hu, Y., et al.: Promptcap: prompt guided image captioning for VQA with GPT-3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2963\u20132975 (2023)"},{"key":"41_CR15","doi-asserted-by":"publisher","first-page":"6132","DOI":"10.18653\/v1\/2024.acl-long.332","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","author":"Q Wang","year":"2024","unstructured":"Wang, Q., et al.: Soft knowledge prompt: Help external knowledge become a better teacher to instruct LLM in knowledge-based VQA. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 6132\u20136143 (2024)"},{"key":"41_CR16","first-page":"3195","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Marino","year":"2019","unstructured":"Marino, K., Rastegari, M., Farhadi, A., Mottaghi, R.: Ok-VQA: a visual question answering benchmark requiring external knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3195\u20133204 (2019)"},{"key":"41_CR17","first-page":"740","volume-title":"European Conference on Computer Vision","author":"TY Lin","year":"2014","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: European Conference on Computer Vision, pp. 740\u2013755. Springer (2014)"},{"key":"41_CR18","doi-asserted-by":"publisher","first-page":"489","DOI":"10.18653\/v1\/2020.findings-emnlp.44","volume-title":"Findings of the Association for Computational Linguistics: EMNLP 2020","author":"F Gard\u00e8res","year":"2020","unstructured":"Gard\u00e8res, F., Ziaeefard, M., Abeloos, B., Lecue, F.: Conceptbert: concept-aware representation for visual question answering. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 489\u2013498 (2020)"},{"key":"41_CR19","first-page":"14111","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Marino","year":"2021","unstructured":"Marino, K., Chen, X., Parikh, D., Gupta, A., Rohrbach, M.: Krisp: integrating implicit and symbolic knowledge for open-domain knowledge-based VQA. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14111\u201314121 (2021)"},{"key":"41_CR20","doi-asserted-by":"publisher","first-page":"4065","DOI":"10.1145\/3581783.3612516","volume-title":"Proceedings of the 31st ACM International Conference on Multimedia","author":"Z Sun","year":"2023","unstructured":"Sun, Z., et al.: Breaking the barrier between pre-training and fine-tuning: a hybrid prompting model for knowledge based VQA. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 4065\u20134073 (2023)"},{"key":"41_CR21","doi-asserted-by":"publisher","first-page":"4030","DOI":"10.1109\/SMC54092.2024.10832004","volume-title":"2024 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","author":"Y Yang","year":"2024","unstructured":"Yang, Y., et al.: Enhancing GPT-3.5 for knowledge-based VQA with in-context prompt learning and image captioning. In: 2024 IEEE International Conference on Systems, Man, and Cybernetics (SMC), pp. 4030\u20134035 (2024)"},{"key":"41_CR22","doi-asserted-by":"publisher","first-page":"10560","DOI":"10.52202\/068431-0767","volume":"35","author":"Y Lin","year":"2022","unstructured":"Lin, Y., et al.: Revive: regional visual representation matters in knowledge-based visual question answering. Adv. Neural Inf. Proces. Syst. 35, 10560\u201310571 (2022)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"41_CR23","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490. (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"41_CR24","first-page":"662","volume-title":"European Conference on Computer Vision","author":"A Kamath","year":"2022","unstructured":"Kamath, A., et al.: Webly supervised concept expansion for general purpose vision models. In: European Conference on Computer Vision, pp. 662\u2013681. Springer (2022)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3507-0_41","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T19:35:13Z","timestamp":1784144113000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3507-0_41"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,16]]},"ISBN":["9789819235063","9789819235070"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3507-0_41","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,16]]},"assertion":[{"value":"16 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}