{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T22:23:08Z","timestamp":1783635788331,"version":"3.55.0"},"reference-count":58,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["24511104000"],"award-info":[{"award-number":["24511104000"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012659","name":"Foundation for Innovative Research Groups of the National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012659","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406073"],"award-info":[{"award-number":["62406073"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.knosys.2026.116358","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:10:59Z","timestamp":1780445459000},"page":"116358","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Evidence-chain-driven multimodal retrieval question answering"],"prefix":"10.1016","volume":"348","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7728-162X","authenticated-orcid":false,"given":"Shuwen","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anran","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9146-051X","authenticated-orcid":false,"given":"Xingjiao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0691-5741","authenticated-orcid":false,"given":"Jiabao","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linlin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liang","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116358_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.envsoft.2020.104828","article-title":"A national scale big data analytics pipeline to assess the potential impacts of flooding on critical infrastructures and communities","volume":"133","author":"Donratanapat","year":"2020","journal-title":"Environ. Model. Softw."},{"key":"10.1016\/j.knosys.2026.116358_b2","doi-asserted-by":"crossref","unstructured":"S. Antol, A. Agrawal, J. Lu, M. Mitchell, D. Batra, C.L. Zitnick, D. Parikh, Vqa: Visual question answering, in: International Conference on Computer Vision, ICCV, 2015, pp. 2425\u20132433.","DOI":"10.1109\/ICCV.2015.279"},{"key":"10.1016\/j.knosys.2026.116358_b3","doi-asserted-by":"crossref","unstructured":"P. Rajpurkar, J. Zhang, K. Lopyrev, P. Liang, SQuAD: 100,000+ Questions for Machine Comprehension of Text, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2016, pp. 2383\u20132392.","DOI":"10.18653\/v1\/D16-1264"},{"key":"10.1016\/j.knosys.2026.116358_b4","article-title":"Answering knowledge-based visual questions via the exploration of question purpose","volume":"133","author":"Song","year":"2023","journal-title":"PR"},{"key":"10.1016\/j.knosys.2026.116358_b5","doi-asserted-by":"crossref","unstructured":"P. Rajpurkar, R. Jia, P. Liang, Know What You Don\u2019t Know: Unanswerable Questions for SQuAD, in: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), 2018, pp. 784\u2013789.","DOI":"10.18653\/v1\/P18-2124"},{"key":"10.1016\/j.knosys.2026.116358_b6","article-title":"Knowledge base graph embedding module design for visual question answering model","volume":"120","author":"Zheng","year":"2021","journal-title":"PR"},{"key":"10.1016\/j.knosys.2026.116358_b7","article-title":"Cross-modal knowledge reasoning for knowledge-based visual question answering","volume":"108","author":"Yu","year":"2020","journal-title":"PR"},{"key":"10.1016\/j.knosys.2026.116358_b8","doi-asserted-by":"crossref","unstructured":"Y. Chang, M. Narang, H. Suzuki, G. Cao, J. Gao, Y. Bisk, Webqa: Multihop and multimodal qa, in: IEEE Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 16495\u201316504.","DOI":"10.1109\/CVPR52688.2022.01600"},{"key":"10.1016\/j.knosys.2026.116358_b9","series-title":"International Conference on Learning Representations","article-title":"MultiModalQA: complex question answering over text, tables and images","author":"Talmor","year":"2020"},{"key":"10.1016\/j.knosys.2026.116358_b10","doi-asserted-by":"crossref","unstructured":"W. Chen, H. Hu, X. Chen, P. Verga, W. Cohen, MuRAG: Multimodal Retrieval-Augmented Generator for Open Question Answering over Images and Text, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2022, pp. 5558\u20135570.","DOI":"10.18653\/v1\/2022.emnlp-main.375"},{"key":"10.1016\/j.knosys.2026.116358_b11","article-title":"Enhancing multi-modal and multi-hop question answering via structured knowledge and unified retrieval-generation","author":"Yang","year":"2023","journal-title":"ACMMM"},{"key":"10.1016\/j.knosys.2026.116358_b12","doi-asserted-by":"crossref","unstructured":"B. Yu, C. Fu, H. Yu, F. Huang, Y. Li, Unified Language Representation for Question Answering over Text, Tables, and Images, in: Findings of the Association for Computational Linguistics: ACL 2023, 2023, pp. 4756\u20134765.","DOI":"10.18653\/v1\/2023.findings-acl.292"},{"key":"10.1016\/j.knosys.2026.116358_b13","series-title":"Prior Analytics","author":"Smith","year":"1989"},{"key":"10.1016\/j.knosys.2026.116358_b14","doi-asserted-by":"crossref","unstructured":"D. Chen, A. Fisch, J. Weston, A. Bordes, Reading Wikipedia to Answer Open-Domain Questions, in: Annual Meeting of the Association for Computational Linguistics, 2017, pp. 1870\u20131879.","DOI":"10.18653\/v1\/P17-1171"},{"key":"10.1016\/j.knosys.2026.116358_b15","doi-asserted-by":"crossref","unstructured":"E. Voorhees, The TREC-8 Question answering track report, in: Proceedings of the Text Retrieval Conference, TREC, 1999.","DOI":"10.6028\/NIST.SP.500-246.qa-overview"},{"key":"10.1016\/j.knosys.2026.116358_b16","series-title":"End-to-end open-domain question answering with BERTserini","first-page":"72","author":"Yang","year":"2019"},{"key":"10.1016\/j.knosys.2026.116358_b17","doi-asserted-by":"crossref","unstructured":"Y. Nie, S. Wang, M. Bansal, Revealing the Importance of Semantic Retrieval for Machine Reading at Scale, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2019, pp. 2553\u20132566.","DOI":"10.18653\/v1\/D19-1258"},{"key":"10.1016\/j.knosys.2026.116358_b18","doi-asserted-by":"crossref","first-page":"183","DOI":"10.1162\/tacl_a_00309","article-title":"Break it down: A question understanding benchmark","volume":"8","author":"Wolfson","year":"2020","journal-title":"TACL"},{"key":"10.1016\/j.knosys.2026.116358_b19","article-title":"Dynamic dual graph networks for textbook question answering","volume":"139","author":"Wang","year":"2023","journal-title":"PR"},{"key":"10.1016\/j.knosys.2026.116358_b20","article-title":"Self-attention driven adversarial similarity learning network","volume":"105","author":"Gao","year":"2020","journal-title":"PR"},{"key":"10.1016\/j.knosys.2026.116358_b21","doi-asserted-by":"crossref","unstructured":"V. Karpukhin, B. Oguz, S. Min, P. Lewis, L. Wu, S. Edunov, D. Chen, W.-t. Yih, Dense Passage Retrieval for Open-Domain Question Answering, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2020.","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"10.1016\/j.knosys.2026.116358_b22","doi-asserted-by":"crossref","unstructured":"O. Khattab, M. Zaharia, Colbert: Efficient and effective passage search via contextualized late interaction over bert, in: Annual ACM Conference on Research and Development in Information Retrieval, 2020, pp. 39\u201348.","DOI":"10.1145\/3397271.3401075"},{"key":"10.1016\/j.knosys.2026.116358_b23","series-title":"RocketQA: An optimized training approach to dense passage retrieval for open-domain question answering","first-page":"5835","author":"Qu","year":"2021"},{"key":"10.1016\/j.knosys.2026.116358_b24","doi-asserted-by":"crossref","unstructured":"R. Ren, Y. Qu, J. Liu, W.X. Zhao, Q. She, H. Wu, H. Wang, J.-R. Wen, RocketQAv2: A Joint Training Method for Dense Passage Retrieval and Passage Re-ranking, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2021, pp. 2825\u20132835.","DOI":"10.18653\/v1\/2021.emnlp-main.224"},{"key":"10.1016\/j.knosys.2026.116358_b25","series-title":"ColBERTv2: Effective and efficient retrieval via lightweight late interaction","first-page":"3715","author":"Santhanam","year":"2022"},{"key":"10.1016\/j.knosys.2026.116358_b26","unstructured":"A. Asai, K. Hashimoto, H. Hajishirzi, R. Socher, C. Xiong, Learning to Retrieve Reasoning Paths over Wikipedia Graph for Question Answering, in: International Conference on Learning Representations, ICLR, 2019."},{"key":"10.1016\/j.knosys.2026.116358_b27","doi-asserted-by":"crossref","unstructured":"M. Yasunaga, H. Ren, A. Bosselut, P. Liang, J. Leskovec, QA-GNN: Reasoning with language models and knowledge graphs for question answering, in: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 2021, pp. 535\u2013546.","DOI":"10.18653\/v1\/2021.naacl-main.45"},{"key":"10.1016\/j.knosys.2026.116358_b28","series-title":"Knowledge guided text retrieval and reading for open domain question answering","author":"Min","year":"2019"},{"key":"10.1016\/j.knosys.2026.116358_b29","doi-asserted-by":"crossref","unstructured":"Z. Yang, P. Qi, S. Zhang, Y. Bengio, W. Cohen, R. Salakhutdinov, C.D. Manning, HotpotQA: A dataset for diverse, explainable multi-hop question answering, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2018, pp. 2369\u20132380.","DOI":"10.18653\/v1\/D18-1259"},{"key":"10.1016\/j.knosys.2026.116358_b30","doi-asserted-by":"crossref","unstructured":"X. Ho, A.-K.D. Nguyen, S. Sugawara, A. Aizawa, Constructing a multi-hop qa dataset for comprehensive evaluation of reasoning steps, in: Proceedings of the International Conference on Computational Linguistics, 2020, pp. 6609\u20136625.","DOI":"10.18653\/v1\/2020.coling-main.580"},{"key":"10.1016\/j.knosys.2026.116358_b31","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume":"33","author":"Lewis","year":"2020","journal-title":"NIPS"},{"key":"10.1016\/j.knosys.2026.116358_b32","series-title":"Proceedings of International Conference on Machine Learning","first-page":"3929","article-title":"Retrieval augmented language model pre-training","author":"Guu","year":"2020"},{"key":"10.1016\/j.knosys.2026.116358_b33","unstructured":"S. Yao, J. Zhao, D. Yu, N. Du, I. Shafran, K. Narasimhan, Y. Cao, ReAct: Synergizing Reasoning and Acting in Language Models, in: International Conference on Learning Representations, ICLR, 2023."},{"key":"10.1016\/j.knosys.2026.116358_b34","doi-asserted-by":"crossref","unstructured":"H. Trivedi, N. Balasubramanian, T. Khot, A. Sabharwal, Interleaving retrieval with chain-of-thought reasoning for knowledge-intensive multi-step questions, in: Annual Meeting of the Association for Computational Linguistics, 2023, pp. 10014\u201310037.","DOI":"10.18653\/v1\/2023.acl-long.557"},{"key":"10.1016\/j.knosys.2026.116358_b35","first-page":"9112","article-title":"Self-rag: Learning to retrieve, generate, and critique through self-reflection","volume":"vol. 2024","author":"Asai","year":"2024"},{"key":"10.1016\/j.knosys.2026.116358_b36","first-page":"7879","article-title":"Manymodalqa: Modality disambiguation and qa over diverse inputs","volume":"vol. 34","author":"Hannan","year":"2020"},{"key":"10.1016\/j.knosys.2026.116358_b37","series-title":"Re2G: Retrieve, rerank, generate","author":"Glass","year":"2022"},{"key":"10.1016\/j.knosys.2026.116358_b38","first-page":"13041","article-title":"Unified vision-language pre-training for image captioning and vqa","volume":"vol. 34","author":"Zhou","year":"2020"},{"key":"10.1016\/j.knosys.2026.116358_b39","series-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","first-page":"4171","author":"Kenton","year":"2019"},{"key":"10.1016\/j.knosys.2026.116358_b40","article-title":"Imagenet classification with deep convolutional neural networks","volume":"25","author":"Krizhevsky","year":"2012","journal-title":"NIPS"},{"key":"10.1016\/j.knosys.2026.116358_b41","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, et al., An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale, in: International Conference on Learning Representations, ICLR, 2020."},{"key":"10.1016\/j.knosys.2026.116358_b42","doi-asserted-by":"crossref","unstructured":"R. Girshick, Fast r-cnn, in: International Conference on Computer Vision, ICCV, 2015, pp. 1440\u20131448.","DOI":"10.1109\/ICCV.2015.169"},{"key":"10.1016\/j.knosys.2026.116358_b43","unstructured":"P. Wang, A. Yang, R. Men, J. Lin, S. Bai, Z. Li, J. Ma, C. Zhou, J. Zhou, H. Yang, Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework, in: Proceedings of International Conference on Machine Learning, ICML, 2022, pp. 23318\u201323340."},{"key":"10.1016\/j.knosys.2026.116358_b44","doi-asserted-by":"crossref","unstructured":"C. Li, H. Xu, J. Tian, W. Wang, M. Yan, B. Bi, J. Ye, H. Chen, G. Xu, Z. Cao, et al., mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections, in: Conference on Empirical Methods in Natural Language Processing, EMNLP, 2022, pp. 7241\u20137259.","DOI":"10.18653\/v1\/2022.emnlp-main.488"},{"key":"10.1016\/j.knosys.2026.116358_b45","series-title":"Mplug-owl: Modularization empowers large language models with multimodality","author":"Ye","year":"2023"},{"key":"10.1016\/j.knosys.2026.116358_b46","unstructured":"E.J. Hu, P. Wallis, Z. Allen-Zhu, Y. Li, S. Wang, L. Wang, W. Chen, et al., LoRA: Low-Rank Adaptation of Large Language Models, in: International Conference on Learning Representations, ICLR, 2021."},{"key":"10.1016\/j.knosys.2026.116358_b47","unstructured":"A. Radford, J.W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, et al., Learning transferable visual models from natural language supervision, in: Proceedings of International Conference on Machine Learning, ICML, 2021, pp. 8748\u20138763."},{"key":"10.1016\/j.knosys.2026.116358_b48","first-page":"27263","article-title":"Bartscore: Evaluating generated text as text generation","volume":"34","author":"Yuan","year":"2021","journal-title":"NIPS"},{"key":"10.1016\/j.knosys.2026.116358_b49","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.knosys.2026.116358_b50","article-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume":"32","author":"Lu","year":"2019","journal-title":"NIPS"},{"key":"10.1016\/j.knosys.2026.116358_b51","series-title":"Annual Meeting of the Association for Computational Linguistics","first-page":"7871","article-title":"BART: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension","author":"Lewis","year":"2020"},{"issue":"4","key":"10.1016\/j.knosys.2026.116358_b52","first-page":"333","article-title":"The probabilistic relevance framework: BM25 and beyond","volume":"3","author":"Robertson","year":"2009","journal-title":"Found. Trends Inf. Retr."},{"key":"10.1016\/j.knosys.2026.116358_b53","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"2318","article-title":"M3-embedding: Multi-linguality, multi-functionality, multi-granularity text embeddings through self-knowledge distillation","author":"Chen","year":"2024"},{"key":"10.1016\/j.knosys.2026.116358_b54","series-title":"Qwen3 technical report","author":"Yang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116358_b55","series-title":"Gemma: Open models based on gemini research and technology","author":"Team","year":"2024"},{"key":"10.1016\/j.knosys.2026.116358_b56","unstructured":"I. Loshchilov, F. Hutter, Decoupled Weight Decay Regularization, in: International Conference on Learning Representations, ICLR, 2018."},{"key":"10.1016\/j.knosys.2026.116358_b57","series-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"10.1016\/j.knosys.2026.116358_b58","series-title":"Constitutional ai: Harmlessness from ai feedback","author":"Bai","year":"2022"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010841?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010841?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T22:09:28Z","timestamp":1783634968000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126010841"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":58,"alternative-id":["S0950705126010841"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116358","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Evidence-chain-driven multimodal retrieval question answering","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116358","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116358"}}