{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T00:19:16Z","timestamp":1783124356044,"version":"3.54.6"},"reference-count":40,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100012542","name":"Sichuan Provincial Science and Technology Support Program","doi-asserted-by":"publisher","award":["2025JDDQ0008"],"award-info":[{"award-number":["2025JDDQ0008"]}],"id":[{"id":"10.13039\/100012542","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100012542","name":"Sichuan Provincial Science and Technology Support Program","doi-asserted-by":"publisher","award":["2024NSFSC0004"],"award-info":[{"award-number":["2024NSFSC0004"]}],"id":[{"id":"10.13039\/100012542","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012406","name":"Organization Department of Sichuan Provincial Party Committee","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012406","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2023YFB3308601"],"award-info":[{"award-number":["2023YFB3308601"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002367","name":"Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["XDA0480401"],"award-info":[{"award-number":["XDA0480401"]}],"id":[{"id":"10.13039\/501100002367","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113953","type":"journal-article","created":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T02:17:51Z","timestamp":1778811471000},"page":"113953","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["DICE: Disentangling Causal Evidence for multimodal textbook question answering via attentive embedding fusion"],"prefix":"10.1016","volume":"179","author":[{"given":"Xu","family":"Gu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingke","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinqiao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaolin","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113953_b1","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, Honolulu, HI, USA, July 21-26, 2017","first-page":"5376","article-title":"Are you smarter than a sixth grader? Textbook question answering for multimodal machine comprehension","author":"Kembhavi","year":"2017"},{"key":"10.1016\/j.patcog.2026.113953_b2","doi-asserted-by":"crossref","unstructured":"D. Caffagni, F. Cocchi, et al., Wiki-llava: Hierarchical retrieval-augmented generation for multimodal llms, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 1818\u20131826.","DOI":"10.1109\/CVPRW63382.2024.00188"},{"key":"10.1016\/j.patcog.2026.113953_b3","series-title":"Learn to Explain: Multimodal Reasoning Via Thought Chains for Science Question Answering","author":"Lu","year":"2022"},{"key":"10.1016\/j.patcog.2026.113953_b4","series-title":"Computer Vision - ECCV 2016 - 14th European Conference, Amsterdam, the Netherlands, October 11-14, 2016, Proceedings, Part IV","first-page":"235","article-title":"A diagram is worth a dozen images","volume":"vol. 9908","author":"Kembhavi","year":"2016"},{"key":"10.1016\/j.patcog.2026.113953_b5","series-title":"A survey of multimodal retrieval-augmented generation","author":"Mei","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b6","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","article-title":"MMed-RAG: Versatile multimodal RAG system for medical vision language models","author":"Xia","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b7","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, EMNLP 2024, Miami, FL, USA, November 12-16, 2024","first-page":"1490","article-title":"UniFashion: A unified vision-language model for multimodal fashion retrieval and generation","author":"Zhao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b8","doi-asserted-by":"crossref","DOI":"10.1016\/j.ins.2025.122704","article-title":"3WD-DRT: A three-way decision enhanced dynamic routing transformer for cost-sensitive multimodal sentiment analysis","volume":"725","author":"Jiang","year":"2026","journal-title":"Inf. Sci."},{"issue":"9","key":"10.1016\/j.patcog.2026.113953_b9","doi-asserted-by":"crossref","first-page":"11872","DOI":"10.1109\/TNNLS.2024.3385436","article-title":"Relation-aware heterogeneous graph network for learning intermodal semantics in textbook question answering","volume":"35","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.113953_b10","article-title":"LLaVA-OneVision: Easy visual task transfer","volume":"2025","author":"Li","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.113953_b11","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113953_b12","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, et al., Improved baselines with visual instruction tuning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"10.1016\/j.patcog.2026.113953_b13","series-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111332","article-title":"Enhancing textual textbook question answering with large language models and retrieval augmented generation","volume":"162","author":"Alawwad","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113953_b15","unstructured":"P. BehnamGhader, V. Adlakha, et al., LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders, in: First Conference on Language Modeling."},{"issue":"2","key":"10.1016\/j.patcog.2026.113953_b16","doi-asserted-by":"crossref","first-page":"42:1","DOI":"10.1145\/3703155","article-title":"A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions","volume":"43","author":"Huang","year":"2025","journal-title":"ACM Trans. Inf. Syst."},{"key":"10.1016\/j.patcog.2026.113953_b17","series-title":"E5-V: universal embeddings with multimodal large language models","author":"Jiang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b18","series-title":"Proceedings of the 35th International Conference on Machine Learning, ICML 2018, Stockholmsm\u00c4ssan, Stockholm, Sweden, July 10-15, 2018","first-page":"530","article-title":"Mutual information neural estimation","author":"Belghazi","year":"2018"},{"key":"10.1016\/j.patcog.2026.113953_b19","series-title":"Proceedings of the 6th Workshop on Gender Bias in Natural Language Processing","first-page":"393","article-title":"Disentangling biased representations: A causal intervention framework for fairer NLP models","author":"Qian","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b20","series-title":"GPT-4 technical report","author":"OpenAI","year":"2023"},{"key":"10.1016\/j.patcog.2026.113953_b21","series-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b22","series-title":"GLM-4.5: agentic, reasoning, and coding (ARC) foundation models","author":"Zeng","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b23","series-title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","author":"Guo","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b24","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024, Miami, Florida, USA, November 12-16, 2024","first-page":"3182","article-title":"Scaling sentence embeddings with large language models","author":"Jiang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b25","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2024, Bangkok, Thailand, August 11-16, 2024","first-page":"10141","article-title":"Meta-task prompting elicits embeddings from large language models","author":"Lei","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b26","series-title":"Robotics: Science and Systems XX, Delft, the Netherlands, July 15-19, 2024","article-title":"RAG-driver: Generalisable driving explanations with retrieval-augmented in-context multi-modal large language model learning","author":"Yuan","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b27","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL 2025 - Volume 1: Long Papers, Albuquerque, New Mexico, USA, April 29 - May 4, 2025","first-page":"6088","article-title":"VisDoM: Multi-document QA with visually rich elements using multimodal retrieval-augmented generation","author":"Suri","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b28","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: EMNLP 2024 - Industry Track, Miami, Florida, USA, November 12-16, 2024","first-page":"1001","article-title":"OMG-QA: building open-domain multi-modal generative question answering systems","author":"Nan","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b29","series-title":"Proceedings of the 31st International Conference on Computational Linguistics, COLING 2025 - System Demonstrations, Abu Dhabi, UAE, January 19-24, 2025","first-page":"126","article-title":"MuRAR: A simple and effective multimodal retrieval and answer refinement framework for multimodal question answering","author":"Zhu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113953_b30","series-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing","first-page":"5469","article-title":"ISAAQ - mastering textbook questions with pre-trained transformers and bottom-up and top-down attention","author":"Gomez-Perez","year":"2020"},{"issue":"5","key":"10.1016\/j.patcog.2026.113953_b31","doi-asserted-by":"crossref","first-page":"1578","DOI":"10.1007\/s11263-023-01954-z","article-title":"Diagram perception networks for textbook question answering via joint optimization","volume":"132","author":"Ma","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.patcog.2026.113953_b32","unstructured":"M. He, A. Zhou, X. Shi, Enhancing textbook question answering with knowledge graph-augmented large language models, in: The 16th Asian Conference on Machine Learning (Conference Track), 2024."},{"key":"10.1016\/j.patcog.2026.113953_b33","series-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, EMNLP 2022, Abu Dhabi, United Arab Emirates, December 7-11, 2022","first-page":"8826","article-title":"PromptBERT: Improving BERT sentence embeddings with prompts","author":"Jiang","year":"2022"},{"key":"10.1016\/j.patcog.2026.113953_b34","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"Language is not all you need: aligning perception with language models","author":"Huang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113953_b35","series-title":"TinyLLaVA: A framework of small-scale large multimodal models","author":"Zhou","year":"2024"},{"key":"10.1016\/j.patcog.2026.113953_b36","series-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"vol. 139","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113953_b37","article-title":"The faiss library","author":"Douze","year":"2025","journal-title":"IEEE Trans. Big Data"},{"key":"10.1016\/j.patcog.2026.113953_b38","series-title":"Proceedings of the the 17th International Workshop on Semantic Evaluation, SemEval@ACL 2023, Toronto, Canada, 13-14 July 2023","first-page":"1906","article-title":"Jack-flood at SemEval-2023 task 5: Hierarchical encoding and reciprocal rank fusion-based system for spoiler classification and generation","author":"Kumar","year":"2023"},{"key":"10.1016\/j.patcog.2026.113953_b39","series-title":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL 2022, Seattle, WA, United States, July 10-15, 2022","first-page":"3715","article-title":"ColBERTv2: Effective and efficient retrieval via lightweight late interaction","author":"Santhanam","year":"2022"},{"key":"10.1016\/j.patcog.2026.113953_b40","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2025, Vienna, Austria, July 27 - August 1, 2025","first-page":"19076","article-title":"MegaPairs: Massive data synthesis for universal multimodal retrieval","author":"Zhou","year":"2025"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326009180?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326009180?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T23:33:07Z","timestamp":1783121587000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326009180"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":40,"alternative-id":["S0031320326009180"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113953","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DICE: Disentangling Causal Evidence for multimodal textbook question answering via attentive embedding fusion","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113953","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113953"}}