{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T01:03:56Z","timestamp":1781139836504,"version":"3.54.1"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100010829","name":"Department of Education of the Xinjiang Uyghur Autonomous Region","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010829","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133104","type":"journal-article","created":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T15:33:43Z","timestamp":1780760023000},"page":"133104","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["R3-RAG: Reweight, rerank, and reflect for evidence-calibrated scientific multimodal reasoning"],"prefix":"10.1016","volume":"331","author":[{"given":"Zihao","family":"Yin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4713-3080","authenticated-orcid":false,"given":"Tao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6564-4745","authenticated-orcid":false,"given":"Yurong","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1617-5804","authenticated-orcid":false,"given":"Kai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ping","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133104_bib0001","series-title":"Findings of the association for computational linguistics: ACL 2025","first-page":"16776","article-title":"Ask in any modality: A comprehensive survey on multimodal retrieval-augmented generation","author":"Abootorabi","year":"2025"},{"issue":"10","key":"10.1016\/j.eswa.2026.133104_bib0002","doi-asserted-by":"crossref","first-page":"952","DOI":"10.1038\/s43588-025-00836-3","article-title":"Probing the limitations of multimodal language models for chemistry and materials research","volume":"5","author":"Alampara","year":"2025","journal-title":"Nature Computational Science"},{"key":"10.1016\/j.eswa.2026.133104_bib0003","series-title":"The twelfth international conference on learning representations (ICLR)","article-title":"Self-RAG: Learning to retrieve, generate, and critique through self-reflection","author":"Asai","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0004","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J., Zhong, H., Zhu, Y., Yang, M., Li, Z., Wan, J., Wang, P., Ding, W., Fu, Z., Xu, Y., Ye, J., Zhang, X., Xie, T., Cheng, Z., Zhang, H., Yang, Z., Xu, H., & Lin, J. (2025). Qwen2.5-VL technical report. arXiv: 2502.13923."},{"key":"10.1016\/j.eswa.2026.133104_bib0005","series-title":"Proceedings of the thirty-eighth AAAI conference on artificial intelligence (AAAI)","first-page":"17682","article-title":"Graph of thoughts: Solving elaborate problems with large language models","author":"Besta","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0006","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"16474","article-title":"WebQA: Multihop and multimodal QA","author":"Chang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133104_bib0007","series-title":"Findings of the association for computational linguistics: ACL 2024","first-page":"2318","article-title":"M3-embedding: Multi-linguality, multi-functionality, multi-granularity text embeddings through self-knowledge distillation","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0008","series-title":"Proceedings of the 2022 conference on empirical methods in natural language processing (EMNLP)","first-page":"5558","article-title":"MuRAG: Multimodal retrieval-augmented generator for open question answering over images and text","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.133104_bib0009","series-title":"Findings of the association for computational linguistics: EMNLP 2025","first-page":"8140","article-title":"VLM is a strong reranker: Advancing multimodal retrieval-augmented generation via knowledge-enhanced reranking and noise-injected training","author":"Chen","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0010","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2591","article-title":"MASS: Overcoming language bias in image-text matching","volume":"vol. 39","author":"Chung","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0011","series-title":"Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing (EMNLP-IJCNLP)","first-page":"55","article-title":"How contextual are contextualized word representations? Comparing the geometry of BERT, ELMo, and GPT-2 embeddings","author":"Ethayarajh","year":"2019"},{"key":"10.1016\/j.eswa.2026.133104_bib0012","series-title":"The thirteenth international conference on learning representations (ICLR)","article-title":"ColPali: Efficient document retrieval with vision language models","author":"Faysse","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0013","series-title":"Proceedings of the 2021 conference on empirical methods in natural language processing (EMNLP)","first-page":"6894","article-title":"SimCSE: Simple contrastive learning of sentence embeddings","author":"Gao","year":"2021"},{"key":"10.1016\/j.eswa.2026.133104_bib0014","unstructured":"Gao, Y., Xiong, Y., Gao, X., Jia, K., Pan, J., Bi, Y., Dai, Y., Sun, J., Guo, Q., Wang, M., & Wang, H. (2023). Retrieval-augmented generation for large language models: A survey. arXiv:abs\/2312.10997. https:\/\/api.semanticscholar.org\/CorpusID:266359151."},{"key":"10.1016\/j.eswa.2026.133104_bib0015","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"12096","article-title":"Language guided visual question answering: Elevate your multimodal language model using knowledge-enriched prompts","author":"Ghosal","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0016","series-title":"Proceedings of the 37th international conference on machine learning","article-title":"Realm: Retrieval-augmented language model pre-training","author":"Guu","year":"2020"},{"key":"10.1016\/j.eswa.2026.133104_bib0017","series-title":"Proceedings of the 1st workshop on NLP for science (NLP4science)","first-page":"58","article-title":"SCITUNE: Aligning large language models with human-curated scientific multimodal instructions","author":"Horawalavithana","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0018","unstructured":"Hsieh, C.-Y., Chen, S.-A., Li, C.-L., Fujii, Y., Ratner, A., Lee, C.-Y., Krishna, R., & Pfister, T. (2023). Tool documentation enables zero-shot tool-usage with large language models. arXiv: 2308.00675."},{"key":"10.1016\/j.eswa.2026.133104_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"23369","article-title":"REVEAL: Retrieval-augmented visual-language pre-training with multi-source multimodal knowledge memory","author":"Hu","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.133104_bib0020","doi-asserted-by":"crossref","DOI":"10.1145\/3571730","article-title":"Survey of hallucination in natural language generation","volume":"55","author":"Ji","year":"2023","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.133104_bib0021","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"13700","article-title":"Chat-UniVi: Unified visual representation empowers large language models with image and video understanding","author":"Jin","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0022","series-title":"Proceedings of the 34th international conference on neural information processing systems","first-page":"793","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","author":"Lewis","year":"2020"},{"key":"10.1016\/j.eswa.2026.133104_bib0023","series-title":"Proceedings of the 40th international conference on machine learning (ICML)","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0024","series-title":"Proceedings of the 63rd annual meeting of the association for computational linguistics (ACL)","first-page":"14163","article-title":"UniRAG: Unified query understanding method for retrieval augmented generation","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0025","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing (EMNLP)","first-page":"292","article-title":"Evaluating object hallucination in large vision-language models","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0026","series-title":"Findings of the association for computational linguistics: EMNLP 2025","first-page":"10276","article-title":"R3-RAG: Learning step-by-step reasoning and retrieval for LLMs via reinforcement learning","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0027","series-title":"Proceedings of the 36th international conference on neural information processing systems","article-title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","author":"Liang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133104_bib0028","unstructured":"Ling, Z., Guo, Z., Huang, Y., An, Y., Xiao, S., Lan, J., Zhu, X., & Zheng, B. (2025). Mmkb-rag: A multi-modal knowledge-based retrieval-augmented generation framework. arXiv: 2504.10074."},{"key":"10.1016\/j.eswa.2026.133104_bib0029","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., & Lee, Y. J. (2024). LLaVA-NeXT: Improved reasoning, ocr, and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/."},{"key":"10.1016\/j.eswa.2026.133104_bib0030","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"2781","article-title":"Hm-rag: Hierarchical multi-agent multimodal retrieval augmented generation","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0031","unstructured":"Lu, H., Liu, W., Zhang, B., Wang, B., Dong, K., Liu, B., Sun, J., Ren, T., Li, Z., Yang, H., Sun, Y., Deng, C., Xu, H., Xie, Z., & Ruan, C. (2024). DeepSeek-VL: Towards real-world vision-language understanding. arXiv: 2403.05525."},{"key":"10.1016\/j.eswa.2026.133104_bib0032","series-title":"Proceedings of the 36th international conference on neural information processing systems","article-title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","author":"Lu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133104_bib0033","series-title":"Advances in neural information processing systems (neurIPS)","first-page":"49326","article-title":"Chameleon: Plug-and-play compositional reasoning with large language models","volume":"vol. 36","author":"Lu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0034","series-title":"Advances in neural information processing systems (neurIPS)","article-title":"Cheap and quick: Efficient vision-language instruction tuning for large language models","volume":"vol. 36","author":"Luo","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0035","series-title":"Proceedings of the 37th international conference on neural information processing systems","article-title":"Self-refine: Iterative refinement with self-feedback","author":"Madaan","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0036","unstructured":"Mei, L., Mo, S., Yang, Z., & Chen, C. (2025). A survey of multimodal retrieval-augmented generation. arXiv: 2504.08748."},{"key":"10.1016\/j.eswa.2026.133104_bib0037","series-title":"Proceedings of the 38th international conference on machine learning (ICML)","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"vol. 139","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133104_bib0038","series-title":"Proceedings of the 37th international conference on neural information processing systems","article-title":"Reflexion: Language agents with verbal reinforcement learning","author":"Shinn","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0039","doi-asserted-by":"crossref","unstructured":"Yan, S.-Q., Gu, J.-C., Zhu, Y., & Ling, Z.-H. (2024). Corrective retrieval augmented generation. OpenReview Preprint. https:\/\/openreview.net\/forum?id=JnWJbrnaUE.","DOI":"10.2139\/ssrn.5267341"},{"key":"10.1016\/j.eswa.2026.133104_bib0040","unstructured":"Yao, F., Tian, C., Liu, J., Zhang, Z., Liu, Q., Jin, L., Li, S., Li, X., & Sun, X. (2023a). Thinking like an expert: Multimodal hypergraph-of-thought (HoT) reasoning to boost foundation models. 10.48550\/arXiv.2308.06207."},{"key":"10.1016\/j.eswa.2026.133104_bib0041","series-title":"Advances in neural information processing systems (neurIPS)","first-page":"11809","article-title":"Tree of thoughts: Deliberate problem solving with large language models","volume":"vol. 36","author":"Yao","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0042","series-title":"The eleventh international conference on learning representations (ICLR)","article-title":"ReAct: Synergizing reasoning and acting in language models","author":"Yao","year":"2023"},{"key":"10.1016\/j.eswa.2026.133104_bib0043","series-title":"Proceedings of the 40th international conference on machine learning (ICML)","first-page":"39755","article-title":"Retrieval-augmented multimodal language modeling","volume":"vol. 202","author":"Yasunaga","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.133104_bib0044","doi-asserted-by":"crossref","DOI":"10.1093\/nsr\/nwae403","article-title":"A survey on multimodal large language models","volume":"11","author":"Yin","year":"2024","journal-title":"National Science Review"},{"key":"10.1016\/j.eswa.2026.133104_bib0045","series-title":"The thirteenth international conference on learning representations (ICLR)","article-title":"VisRAG: Vision-based retrieval-augmented generation on multi-modality documents","author":"Yu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0046","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR)","first-page":"9556","article-title":"MMMU: A massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI","author":"Yue","year":"2024"},{"key":"10.1016\/j.eswa.2026.133104_bib0047","series-title":"Proceedings of the 2025 international conference on content-Based multimedia indexing (CBMI)","first-page":"1","article-title":"ELIP: Enhanced visual-language foundation models for image retrieval","author":"Zhan","year":"2025"},{"key":"10.1016\/j.eswa.2026.133104_bib0048","series-title":"Advances in neural information processing systems (neurIPS)","first-page":"5168","article-title":"DDCoT: Duty-distinct chain-of-thought prompting for multimodal reasoning in language models","volume":"vol. 36","author":"Zheng","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426020154?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426020154?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T00:44:16Z","timestamp":1781138656000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426020154"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":48,"alternative-id":["S0957417426020154"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133104","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"R3-RAG: Reweight, rerank, and reflect for evidence-calibrated scientific multimodal reasoning","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133104","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133104"}}