{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T18:24:02Z","timestamp":1786991042273,"version":"build-2736575974"},"reference-count":232,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T00:00:00Z","timestamp":1776816000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Fusion"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.inffus.2026.104405","type":"journal-article","created":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T23:07:09Z","timestamp":1776985629000},"page":"104405","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":2,"special_numbering":"C","title":["Decoding the multimodal maze: A systematic review on the adoption of explainability in multimodal attention-based models"],"prefix":"10.1016","volume":"135","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3855-9163","authenticated-orcid":false,"given":"Md Raisul","family":"Kibria","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5286-5343","authenticated-orcid":false,"given":"S\u00e9bastien","family":"Lafond","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2683-3775","authenticated-orcid":false,"given":"Janan","family":"Arslan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"5","key":"10.1016\/j.inffus.2026.104405_bib0001","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3236009","article-title":"A survey of methods for explaining black box models","volume":"51","author":"Guidotti","year":"2018","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.inffus.2026.104405_bib0002","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1016\/j.inffus.2019.12.012","article-title":"Explainable artificial intelligence (XAI): concepts, taxonomies, opportunities and challenges toward responsible AI","volume":"58","author":"Arrieta","year":"2020","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104405_bib0003","doi-asserted-by":"crossref","first-page":"245","DOI":"10.1613\/jair.1.12228","article-title":"A survey on the explainability of supervised machine learning","volume":"70","author":"Burkart","year":"2021","journal-title":"J. Artif. Intell. Res."},{"key":"10.1016\/j.inffus.2026.104405_bib0004","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1016\/j.inffus.2021.07.016","article-title":"Unbox the black-box for the medical explainable AI via multi-modal and multi-centre data fusion: a mini-review, two showcases and beyond","volume":"77","author":"Yang","year":"2022","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104405_bib0005","series-title":"Proceedings of the 2023 ACM Conference on Fairness, Accountability, and Transparency","first-page":"1198","article-title":"Explainability in AI policies: a critical review of communications, reports, regulations, and standards in the EU, US, and UK","author":"Nannini","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0006","doi-asserted-by":"crossref","unstructured":"J. Chun, C.S. de Witt, K. Elkins, Comparative global AI regulation: policy perspectives from the EU, China, and the US, 2024, arXiv: 2410.21279.","DOI":"10.2139\/ssrn.5104429"},{"key":"10.1016\/j.inffus.2026.104405_bib0007","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6\u201312, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.inffus.2026.104405_bib0008","series-title":"Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP","first-page":"353","article-title":"GLUE: a multi-task benchmark and analysis platform for natural language understanding","author":"Wang","year":"2018"},{"key":"10.1016\/j.inffus.2026.104405_bib0009","series-title":"7th Joint International Workshop, CVII-STENT 2018 and Third International Workshop, LABELS 2018, Held in Conjunction with MICCAI 2018, Granada, Spain, September 16, 2018, Proceedings 3","first-page":"180","article-title":"Radiology objects in context (ROCO): a multimodal image dataset","author":"Pelka","year":"2018"},{"key":"10.1016\/j.inffus.2026.104405_bib0010","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"2425","article-title":"Vqa: visual question answering","author":"Antol","year":"2015"},{"key":"10.1016\/j.inffus.2026.104405_bib0011","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int. J. Comput. Vis."},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0012","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1038\/sdata.2016.35","article-title":"MIMIC-III, A freely accessible critical care database","volume":"3","author":"Johnson","year":"2016","journal-title":"Sci. Data"},{"issue":"3","key":"10.1016\/j.inffus.2026.104405_bib0013","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3446374","article-title":"Generative adversarial networks (GANs) challenges, solutions, and future directions","volume":"54","author":"Saxena","year":"2021","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.inffus.2026.104405_bib0014","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1186\/s40537-021-00444-8","article-title":"Review of deep learning: concepts, CNN architectures, challenges, applications, future directions","volume":"8","author":"Alzubaidi","year":"2021","journal-title":"J. Big Data"},{"key":"10.1016\/j.inffus.2026.104405_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2021.101268","article-title":"Combining context-relevant features with multi-stage attention network for short text classification","volume":"71","author":"Liu","year":"2022","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.inffus.2026.104405_bib0016","unstructured":"D. Bahdanau, K. Cho, Y. Bengio, Neural machine translation by jointly learning to align and translate, 2014, arXiv: 1409.0473."},{"key":"10.1016\/j.inffus.2026.104405_sbref0017","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0018","unstructured":"A. Dosovitskiy, An image is worth 16x16 words: transformers for image recognition at scale, 2020, arXiv: 2010.11929."},{"key":"10.1016\/j.inffus.2026.104405_bib0019","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3202","article-title":"Video swin transformer","author":"Liu","year":"2022"},{"issue":"10","key":"10.1016\/j.inffus.2026.104405_bib0020","doi-asserted-by":"crossref","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","article-title":"Multimodal learning with transformers: a survey","volume":"45","author":"Xu","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.inffus.2026.104405_bib0021","unstructured":"S.S. Sengar, A.B. Hasan, S. Kumar, F. Carroll, Generative artificial intelligence: a systematic review and applications, 2024, arXiv: 2405.11029."},{"key":"10.1016\/j.inffus.2026.104405_bib0022","doi-asserted-by":"crossref","unstructured":"S. Abnar, W. Zuidema, Quantifying attention flow in transformers, 2020, arXiv: 2005.00928.","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"10.1016\/j.inffus.2026.104405_bib0023","unstructured":"Y. Qiang, D. Pan, C. Li, X. Li, R. Jang, D. Zhu, et al., AttCAT: explaining transformers via attentive class activation tokens, 2026."},{"key":"10.1016\/j.inffus.2026.104405_bib0024","unstructured":"L. Parcalabescu, A. Frank, Mm-shap: a performance-agnostic metric for measuring multimodal contributions in vision and language models & tasks, 2022, arXiv: 2212.08158."},{"key":"10.1016\/j.inffus.2026.104405_bib0025","doi-asserted-by":"crossref","first-page":"159794","DOI":"10.1109\/ACCESS.2024.3467062","article-title":"Multimodal explainable artificial intelligence: a comprehensive review of methodological advances and future research directions","volume":"12","author":"Rodis","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.inffus.2026.104405_bib0026","unstructured":"F. Doshi-Velez, B. Kim, Towards a rigorous science of interpretable machine learning, 2017, arXiv: 1702.08608."},{"key":"10.1016\/j.inffus.2026.104405_bib0027","series-title":"2018 IEEE 5th International Conference on Data Science and Advanced Analytics (DSAA)","first-page":"80","article-title":"Explaining explanations: an overview of interpretability of machine learning","author":"Gilpin","year":"2018"},{"issue":"4","key":"10.1016\/j.inffus.2026.104405_bib0028","doi-asserted-by":"crossref","first-page":"92","DOI":"10.3390\/computers13040092","article-title":"The explainability of transformers: current status and directions","volume":"13","author":"Fantozzi","year":"2024","journal-title":"Computers"},{"issue":"3\u20134","key":"10.1016\/j.inffus.2026.104405_bib0029","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3387166","article-title":"A multidisciplinary survey and framework for design and evaluation of explainable AI systems","volume":"11","author":"Mohseni","year":"2021","journal-title":"ACM Trans. Interact. Intell. Syst."},{"issue":"2004","key":"10.1016\/j.inffus.2026.104405_bib0030","first-page":"1","article-title":"Procedures for performing systematic reviews","volume":"33","author":"Kitchenham","year":"2004","journal-title":"Keele UK Keele Univ."},{"issue":"5","key":"10.1016\/j.inffus.2026.104405_bib0031","first-page":"336","article-title":"Preferred reporting items for systematic reviews and meta-analyses: the PRISMA statement","volume":"8","author":"Moher","year":"2009","journal-title":"Bmj"},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0032","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/j.rbmo.2023.04.009","article-title":"Artificial intelligence in scientific writing: a friend or a foe?","volume":"47","author":"Altm\u00e4e","year":"2023","journal-title":"Reprod. BioMed. Online"},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0033","doi-asserted-by":"crossref","first-page":"1418","DOI":"10.1038\/s41467-024-45563-x","article-title":"Structured information extraction from scientific text with large language models","volume":"15","author":"Dagdelen","year":"2024","journal-title":"Nat. Commun."},{"key":"10.1016\/j.inffus.2026.104405_bib0034","series-title":"Proceedings of the 18th ACM\/IEEE International Symposium on Empirical Software Engineering and Measurement","first-page":"25","article-title":"ChatGPT application in systematic literature reviews in software engineering: an evaluation of its accuracy to support the selection activity","author":"Felizardo","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0035","series-title":"Proceedings of the 28th International Conference on Evaluation and Assessment in Software Engineering","first-page":"262","article-title":"The promise and challenges of using LLMs to accelerate the screening process of systematic reviews","author":"Huotala","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0036","unstructured":"A. Grattafiori, A. Dubey, A. Jauhri, A. Pandey, A. Kadian, A. Al-Dahle, A. Letman, A. Mathur, A. Schelten, A. Vaughan, et al., The llama 3 herd of models, 2024, arXiv: 2407.21783."},{"key":"10.1016\/j.inffus.2026.104405_bib0037","doi-asserted-by":"crossref","first-page":"24824","DOI":"10.52202\/068431-1800","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"35","author":"Wei","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0038","series-title":"Proceedings of the 18th International Conference on Evaluation and Assessment in Software Engineering","first-page":"1","article-title":"Guidelines for snowballing in systematic literature studies and a replication in software engineering","author":"Wohlin","year":"2014"},{"key":"10.1016\/j.inffus.2026.104405_bib0039","series-title":"2021 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"387","article-title":"Generic attention-model explainability for interpreting bi-modal and encoder-decoder transformers","author":"Chefer","year":"2021"},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0040","doi-asserted-by":"crossref","first-page":"366","DOI":"10.1016\/j.ejor.2023.06.024","article-title":"360\u202fDegrees rumor detection: when explanations got some explaining to do","volume":"317","author":"Janssens","year":"2024","journal-title":"Eur. J. Oper. Res."},{"key":"10.1016\/j.inffus.2026.104405_bib0041","doi-asserted-by":"crossref","first-page":"392","DOI":"10.1016\/j.neunet.2022.03.017","article-title":"A BERT based dual-channel explainable text emotion recognition system","volume":"150","author":"Kumar","year":"2022","journal-title":"Neural Netw."},{"key":"10.1016\/j.inffus.2026.104405_bib0042","series-title":"Proceedings of the 2023 International Conference on Computer, Vision and Intelligent Technology","first-page":"1","article-title":"A case-based channel selection method for eeg emotion recognition using interpretable transformer networks","author":"Du","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0043","doi-asserted-by":"crossref","first-page":"191","DOI":"10.1016\/j.jare.2022.08.021","article-title":"A hybrid explainable ensemble transformer encoder for pneumonia identification from chest X-ray images","volume":"48","author":"Ukwuoma","year":"2023","journal-title":"J. Adv. Res."},{"key":"10.1016\/j.inffus.2026.104405_bib0044","series-title":"Database Systems for Advanced Applications","first-page":"336","article-title":"A novel deep learning framework for interpretable drug-target interaction prediction with attention and multi-task mechanism","volume":"13946","author":"Zheng","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0045","doi-asserted-by":"crossref","unstructured":"P. Bhargava, Adaptive Transformers for Learning Multimodal Representations, 2020, 10.48550\/arXiv.2005.07486.","DOI":"10.18653\/v1\/2020.acl-srw.1"},{"key":"10.1016\/j.inffus.2026.104405_bib0046","series-title":"2023 9th International Conference on Big Data and Information Analytics (BigDIA)","first-page":"802","article-title":"An explainable recommendation method based on diffusion model","author":"Guo","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0047","first-page":"1","article-title":"An explainable spatial\u2013frequency multiscale transformer for remote sensing scene classification","volume":"61","author":"Yang","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.inffus.2026.104405_bib0048","doi-asserted-by":"crossref","DOI":"10.3389\/fbinf.2023.1274599","article-title":"Attention network for predicting T-cell receptor\u2013peptide binding can associate attention with interpretable protein structural properties","volume":"3","author":"Koyama","year":"2023","journal-title":"Front. Bioinform."},{"key":"10.1016\/j.inffus.2026.104405_bib0049","doi-asserted-by":"crossref","unstructured":"J. Ferrando, M.R. Costa-juss\u00e0, Attention weights in transformer NMT fail aligning words between sequences but largely explain model predictions, 2021, arXiv: 2109.05853[cs].","DOI":"10.18653\/v1\/2021.findings-emnlp.39"},{"key":"10.1016\/j.inffus.2026.104405_bib0050","unstructured":"M. Rigotti, C. Miksovic, I. Giurgiu, T. Gschwind, P. Scotton, et al., Attention-based interpretability with concept transformersin: International Conference on Learning Representations, 2022https:\/\/openreview.net\/forum?id=kAa9eDS0RdO."},{"key":"10.1016\/j.inffus.2026.104405_bib0051","series-title":"Machine Learning in Clinical Neuroimaging","first-page":"125","article-title":"Augmenting magnetic resonance imaging with tabular features for enhanced and interpretable medial temporal lobe atrophy prediction","volume":"13596","author":"Lee","year":"2022"},{"issue":"8","key":"10.1016\/j.inffus.2026.104405_bib0052","doi-asserted-by":"crossref","first-page":"3121","DOI":"10.1109\/JBHI.2021.3063721","article-title":"Bidirectional representation learning from transformers using multimodal electronic health record data to predict depression","volume":"25","author":"Meng","year":"2021","journal-title":"IEEE J. Biomed. Health Inform."},{"issue":"4","key":"10.1016\/j.inffus.2026.104405_bib0053","doi-asserted-by":"crossref","DOI":"10.1093\/bib\/bbad231","article-title":"DeepSTF: predicting transcription factor binding sites by interpretable deep neural networks combining sequence and shape","volume":"24","author":"Ding","year":"2023","journal-title":"Brief. Bioinform."},{"key":"10.1016\/j.inffus.2026.104405_bib0054","doi-asserted-by":"crossref","unstructured":"M. Feucht, Z. Wu, S. Althammer, V. Tresp, et al., Description-based label attention classifier for explainable ICD-9\u202fclassification, 2021, arXiv: 2109.12026[cs].","DOI":"10.18653\/v1\/2021.wnut-1.8"},{"key":"10.1016\/j.inffus.2026.104405_bib0055","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2023.108460","article-title":"DFYOLOv5m-M2transformer: interpretation of vegetable disease recognition results using image dense captioning techniques","volume":"215","author":"Sun","year":"2023","journal-title":"Comput. Electron. Agric."},{"key":"10.1016\/j.inffus.2026.104405_bib0056","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109666","article-title":"eX-ViT: a novel explainable vision transformer for weakly supervised semantic segmentation","volume":"142","author":"Yu","year":"2023","journal-title":"Pattern Recognit."},{"issue":"11","key":"10.1016\/j.inffus.2026.104405_bib0057","doi-asserted-by":"crossref","first-page":"248","DOI":"10.3390\/jimaging9110248","article-title":"Explainable connectionist-temporal-classification-based scene text recognition","volume":"9","author":"Buoy","year":"2023","journal-title":"J. Imaging"},{"key":"10.1016\/j.inffus.2026.104405_bib0058","series-title":"Proceedings of the 8th International Conference on Computer and Communications Management","first-page":"19","article-title":"Explainable deep learning for thai stock market prediction using textual representation and technical indicators","author":"Chiewhawan","year":"2020"},{"issue":"18","key":"10.1016\/j.inffus.2026.104405_bib0059","doi-asserted-by":"crossref","first-page":"6766","DOI":"10.3390\/s22186766","article-title":"Explainable malware detection system using transformers-based transfer learning and multi-model visual representation","volume":"22","author":"Ullah","year":"2022","journal-title":"Sensors"},{"key":"10.1016\/j.inffus.2026.104405_bib0060","doi-asserted-by":"crossref","DOI":"10.1016\/j.trc.2022.103829","article-title":"Explainable multimodal trajectory prediction using attention models","volume":"143","author":"Zhang","year":"2022","journal-title":"Transp. Res. C: Emerg. Technol."},{"key":"10.1016\/j.inffus.2026.104405_bib0061","series-title":"2023 IEEE 23rd International Working Conference on Source Code Analysis and Manipulation (SCAM)","first-page":"96","article-title":"Explaining transformer-based code models: what do they learn? when they do not work?","author":"Mohammadkhani","year":"2023"},{"issue":"15","key":"10.1016\/j.inffus.2026.104405_bib0062","first-page":"13933","article-title":"Exploring explainable selection to control abstractive summarization","volume":"35","author":"Wang","year":"2021","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"03","key":"10.1016\/j.inffus.2026.104405_bib0063","first-page":"1369","article-title":"Faster, stronger, and more interpretable: massive transformer architectures for vision-language tasks","volume":"03","author":"Chen","year":"2023","journal-title":"Adv. Artif. Intell. Mach. Learn."},{"key":"10.1016\/j.inffus.2026.104405_bib0064","series-title":"ACM Multimedia Asia 2023","first-page":"1","article-title":"Generic attention-model explainability by weighted relevance accumulation","author":"Huang","year":"2023"},{"issue":"11","key":"10.1016\/j.inffus.2026.104405_bib0065","doi-asserted-by":"crossref","first-page":"14493","DOI":"10.1007\/s10489-022-04254-0","article-title":"Interpretable tourism demand forecasting with temporal fusion transformers amid COVID-19","volume":"53","author":"Wu","year":"2023","journal-title":"Appl. Intell."},{"key":"10.1016\/j.inffus.2026.104405_bib0066","doi-asserted-by":"crossref","unstructured":"M. Parelli, D. Mallis, M. Diomataris, V. Pitsikalis, et al., Interpretable visual question answering via reasoning supervision, 2023, arXiv: 2309.03726[cs].","DOI":"10.1109\/ICIP49359.2023.10223156"},{"key":"10.1016\/j.inffus.2026.104405_bib0067","doi-asserted-by":"crossref","unstructured":"I. Malkiel, D. Ginzburg, O. Barkan, A. Caciularu, J. Weill, N. Koenigstein, et al., Interpreting BERT-based text similarity via activation and saliency maps, 2022, arXiv: 2208.06612[cs].","DOI":"10.1145\/3485447.3512045"},{"issue":"4","key":"10.1016\/j.inffus.2026.104405_bib0068","doi-asserted-by":"crossref","first-page":"305","DOI":"10.1007\/s10590-020-09254-w","article-title":"Investigating alignment interpretability for low-resource NMT","volume":"34","author":"Boito","year":"2020","journal-title":"Mach. Transl."},{"key":"10.1016\/j.inffus.2026.104405_bib0069","series-title":"Proceedings of the 2nd Workshop on Evaluation and Comparison of NLP Systems","first-page":"133","article-title":"IST-unbabel 2021\u202fsubmission for the explainable quality estimation shared task","author":"Treviso","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0070","series-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"256","article-title":"KERMIT: complementing transformer architectures with encoders of explicit syntactic interpretations","author":"Zanzotto","year":"2020"},{"key":"10.1016\/j.inffus.2026.104405_bib0071","series-title":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","first-page":"1055","article-title":"Logiformer: a two-branch graph transformer network for interpretable logical reasoning","author":"Xu","year":"2022"},{"key":"10.1016\/j.inffus.2026.104405_bib0072","series-title":"2020 IEEE International Conference on Energy Internet (ICEI)","first-page":"134","article-title":"Multi-granular BERT: an interpretable model applicable to internet-of-thing devices","author":"Xu","year":"2020"},{"key":"10.1016\/j.inffus.2026.104405_bib0073","first-page":"1","article-title":"Multiscale time-frequency sparse transformer based on partly interpretable method for bearing fault diagnosis","volume":"2023","author":"Che","year":"2023","journal-title":"Shock Vib."},{"issue":"18","key":"10.1016\/j.inffus.2026.104405_bib0074","doi-asserted-by":"crossref","first-page":"7875","DOI":"10.3390\/s23187875","article-title":"Natural-language-driven multimodal representation learning for audio-visual scene-aware dialog system","volume":"23","author":"Heo","year":"2023","journal-title":"Sensors"},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0075","doi-asserted-by":"crossref","first-page":"589","DOI":"10.1109\/TNNLS.2020.3027595","article-title":"Neural encoding and decoding with distributed sentence representations","volume":"32","author":"Sun","year":"2021","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0076","doi-asserted-by":"crossref","DOI":"10.1016\/j.compag.2023.107863","article-title":"ODP-Transformer: interpretation of pest classification results using image caption generation techniques","volume":"209","author":"Wang","year":"2023","journal-title":"Comput. Electron. Agric."},{"key":"10.1016\/j.inffus.2026.104405_bib0077","doi-asserted-by":"crossref","unstructured":"X. Li, X. Yin, C. Li, P. Zhang, X. Hu, L. Zhang, L. Wang, H. Hu, L. Dong, F. Wei, Y. Choi, J. Gao, et al., Oscar: object-semantics aligned pre-training for vision-language tasks, 2020, arXiv: 2004.06165[cs].","DOI":"10.1007\/978-3-030-58577-8_8"},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0078","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1007\/s11263-021-01493-5","article-title":"Physical representation learning and parameter identification from video using differentiable physics","volume":"130","author":"Kandukuri","year":"2022","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.inffus.2026.104405_bib0079","doi-asserted-by":"crossref","DOI":"10.1016\/j.jbi.2023.104427","article-title":"Representation of time-varying and time-invariant EMR data and its application in modeling outcome prediction for heart failure patients","volume":"143","author":"Huang","year":"2023","journal-title":"J. Biomed. Inform."},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0080","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3542822","article-title":"Supervised contrastive learning for interpretable long-form document matching","volume":"17","author":"Jha","year":"2023","journal-title":"ACM Trans. Knowl. Discov. Data"},{"issue":"3","key":"10.1016\/j.inffus.2026.104405_bib0081","doi-asserted-by":"crossref","first-page":"782","DOI":"10.1021\/acs.jcim.2c01283","article-title":"TFRegNCI: interpretable noncovalent interaction correction multimodal based on transformer encoder fusion","volume":"63","author":"Wang","year":"2023","journal-title":"J. Chem. Inf. Model."},{"key":"10.1016\/j.inffus.2026.104405_bib0082","doi-asserted-by":"crossref","unstructured":"J. Ferrando, G.I. G\u00e1llego, B. Alastruey, C. Escolano, M.R. Costa-juss\u00e0, et al., Towards opening the black box of neural machine translation: source and target interpretations of the transformer, 2022, arXiv: 2205.11631[cs].","DOI":"10.18653\/v1\/2022.emnlp-main.599"},{"key":"10.1016\/j.inffus.2026.104405_bib0083","series-title":"Interspeech 2021","first-page":"1748","article-title":"Towards the explainability of multimodal speech emotion recognition","author":"Kumar","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0084","doi-asserted-by":"crossref","DOI":"10.1016\/j.media.2023.103040","article-title":"Transformer with convolution and graph-node co-embedding: an accurate and interpretable vision backbone for predicting gene expressions from local histopathological image","volume":"91","author":"Xiao","year":"2024","journal-title":"Med. Image Anal."},{"key":"10.1016\/j.inffus.2026.104405_bib0085","series-title":"Advances in Information Retrieval","first-page":"327","article-title":"Using the hammer only on nails: a hybrid method for representation-based evidence retrieval for question answering","volume":"Vol. 12656","author":"Liang","year":"2021"},{"issue":"4","key":"10.1016\/j.inffus.2026.104405_bib0086","doi-asserted-by":"crossref","first-page":"1681","DOI":"10.1109\/JBHI.2022.3163751","article-title":"Vision-language transformer for interpretable pathology visual question answering","volume":"27","author":"Naseem","year":"2023","journal-title":"IEEE J. Biomed. Health Inform."},{"key":"10.1016\/j.inffus.2026.104405_bib0087","doi-asserted-by":"crossref","DOI":"10.3389\/frai.2021.767971","article-title":"What does a language-and-vision transformer see: the impact of semantic information on visual representations","volume":"4","author":"Ilinykh","year":"2021","journal-title":"Front. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104405_bib0088","doi-asserted-by":"crossref","DOI":"10.1016\/j.trc.2023.104358","article-title":"Why did the AI make that decision? towards an explainable artificial intelligence (XAI) for autonomous driving systems","volume":"156","author":"Dong","year":"2023","journal-title":"Transp. Res. C Emerg. Technol."},{"key":"10.1016\/j.inffus.2026.104405_bib0089","doi-asserted-by":"crossref","unstructured":"F. Lin, M. Li, D. Li, T. Hospedales, Y.-Z. Song, Y. Qi, et al., Zero-shot everything sketch-based image retrieval, and in explainable style, 2023, arXiv: 2303.14348[cs].","DOI":"10.1109\/CVPR52729.2023.02236"},{"key":"10.1016\/j.inffus.2026.104405_bib0090","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations)","first-page":"421","article-title":"Inseq: an interpretability toolkit for sequence generation models","author":"Sarti","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0091","doi-asserted-by":"crossref","unstructured":"S. Katz, Y. Belinkov, VISIT: visualizing and interpreting the semantic information flow of transformers, 2023, arXiv: 2305.13417[cs].","DOI":"10.18653\/v1\/2023.findings-emnlp.939"},{"key":"10.1016\/j.inffus.2026.104405_bib0092","doi-asserted-by":"crossref","unstructured":"E. Aflalo, M. Du, S.-Y. Tseng, Y. Liu, C. Wu, N. Duan, V. Lal, et al., VL-InterpreT: an interactive visualization tool for interpreting vision-language transformers, 2022, arXiv: 2203.17247[cs].","DOI":"10.1109\/CVPR52688.2022.02072"},{"issue":"13s","key":"10.1016\/j.inffus.2026.104405_bib0093","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3583558","article-title":"From anecdotal evidence to quantitative evaluation methods: a systematic review on evaluating explainable AI","volume":"55","author":"Nauta","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.inffus.2026.104405_sbref0094","article-title":"An online sequence-to-sequence model using partial conditioning","volume":"29","author":"Jaitly","year":"2016","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0095","series-title":"Interspeech","first-page":"3702","article-title":"An analysis of \u201cAttention\u201d in sequence-to-sequence models","author":"Prabhavalkar","year":"2017"},{"key":"10.1016\/j.inffus.2026.104405_bib0096","unstructured":"H. Tan, M. Bansal, Lxmert: learning cross-modality encoder representations from transformers, 2019, arXiv: 1908.07490."},{"key":"10.1016\/j.inffus.2026.104405_bib0097","unstructured":"G.M. Correia, V. Niculae, A.F.T. Martins, Adaptively sparse transformers, 2019, arXiv: 1909.00015."},{"issue":"3","key":"10.1016\/j.inffus.2026.104405_bib0098","doi-asserted-by":"crossref","first-page":"615","DOI":"10.3390\/make3030032","article-title":"Classification of explainable artificial intelligence methods through their output formats","volume":"3","author":"Vilone","year":"2021","journal-title":"Mach. Learn. Knowl. Extr."},{"key":"10.1016\/j.inffus.2026.104405_bib0099","series-title":"Natural Language Processing and Chinese Computing: 8th cCF International Conference, NLPCC 2019, Dunhuang, China, October 9\u201314, 2019, Proceedings, Part II 8","first-page":"563","article-title":"Explainable AI: a brief survey on history, research areas, approaches and challenges","author":"Xu","year":"2019"},{"key":"10.1016\/j.inffus.2026.104405_bib0100","unstructured":"S. Jain, B.C. Wallace, Attention is not explanation, 2019, arXiv: 1902.10186."},{"key":"10.1016\/j.inffus.2026.104405_bib0101","series-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics: System Demonstrations","first-page":"37","article-title":"A multiscale visualization of attention in the transformer model","author":"Vig","year":"2019"},{"key":"10.1016\/j.inffus.2026.104405_bib0102","doi-asserted-by":"crossref","unstructured":"J. Ferrando, G.I. G\u00e1llego, M.R. Costa-Juss\u00e0, Measuring the mixing of contextual information in the transformer, 2022, arXiv: 2203.04212.","DOI":"10.18653\/v1\/2022.emnlp-main.595"},{"key":"10.1016\/j.inffus.2026.104405_bib0103","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"618","article-title":"Grad-cam: visual explanations from deep networks via gradient-based localization","author":"Selvaraju","year":"2017"},{"key":"10.1016\/j.inffus.2026.104405_bib0104","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"782","article-title":"Transformer interpretability beyond attention visualization","author":"Chefer","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0105","unstructured":"A. Fan, E. Grave, A. Joulin, Reducing transformer depth on demand with structured dropout, 2019, arXiv: 1909.11556."},{"key":"10.1016\/j.inffus.2026.104405_bib0106","doi-asserted-by":"crossref","unstructured":"M. Hartmann, D. Sonntag, A survey on improving NLP models with human explanations, 2022, arXiv: 2204.08892.","DOI":"10.18653\/v1\/2022.lnls-1.5"},{"issue":"8","key":"10.1016\/j.inffus.2026.104405_bib0107","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"issue":"34","key":"10.1016\/j.inffus.2026.104405_bib0108","first-page":"1","article-title":"Quantus: an explainable ai toolkit for responsible evaluation of neural network explanations and beyond","volume":"24","author":"Hedstr\u00f6m","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104405_bib0109","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1016\/j.inffus.2021.05.009","article-title":"Notions of explainability and evaluation approaches for explainable artificial intelligence","volume":"76","author":"Vilone","year":"2021","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104405_bib0110","series-title":"Trec","first-page":"77","article-title":"The trec-8 question answering track report","volume":"Vol. 99","author":"Voorhees","year":"1999"},{"key":"10.1016\/j.inffus.2026.104405_bib0111","series-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics","first-page":"311","article-title":"Bleu: a method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.inffus.2026.104405_bib0112","series-title":"Text Summarization Branches Out","first-page":"74","article-title":"Rouge: a package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.inffus.2026.104405_bib0113","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4566","article-title":"Cider: consensus-based image description evaluation","author":"Vedantam","year":"2015"},{"key":"10.1016\/j.inffus.2026.104405_bib0114","series-title":"Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part V 14","first-page":"382","article-title":"Spice: semantic propositional image caption evaluation","author":"Anderson","year":"2016"},{"key":"10.1016\/j.inffus.2026.104405_bib0115","series-title":"Proceedings of the Ninth Workshop on Statistical Machine Translation","first-page":"376","article-title":"Meteor universal: language specific translation evaluation for any target language","author":"Denkowski","year":"2014"},{"key":"10.1016\/j.inffus.2026.104405_bib0116","unstructured":"X. Chen, H. Fang, T.-Y. Lin, R. Vedantam, S. Gupta, P. Doll\u00e1r, C.L. Zitnick, Microsoft coco captions: data collection and evaluation server, 2015, arXiv: 1504.00325."},{"issue":"8","key":"10.1016\/j.inffus.2026.104405_bib0117","doi-asserted-by":"crossref","first-page":"832","DOI":"10.3390\/electronics8080832","article-title":"Machine learning interpretability: a survey on methods and metrics","volume":"8","author":"Carvalho","year":"2019","journal-title":"Electronics"},{"key":"10.1016\/j.inffus.2026.104405_bib0118","doi-asserted-by":"crossref","first-page":"765","DOI":"10.1109\/TASLP.2023.3293030","article-title":"Overview of the tenth dialog system technology challenge: DSTC10","volume":"32","author":"Yoshino","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.inffus.2026.104405_bib0119","doi-asserted-by":"crossref","unstructured":"Y. Liu, C. Wu, S.-y. Tseng, V. Lal, X. He, N. Duan, KD-VLP: improving end-to-end vision-and-language pretraining with object knowledge distillation, 2021, arXiv: 2109.10504.","DOI":"10.18653\/v1\/2022.findings-naacl.119"},{"issue":"12","key":"10.1016\/j.inffus.2026.104405_bib0120","doi-asserted-by":"crossref","DOI":"10.1093\/nsr\/nwae403","article-title":"A survey on multimodal large language models","volume":"11","author":"Yin","year":"2024","journal-title":"Natl. Sci. Rev."},{"key":"10.1016\/j.inffus.2026.104405_bib0121","unstructured":"S.N. Wadekar, A. Chaurasia, A. Chadha, E. Culurciello, The evolution of multimodal model architectures, 2024, arXiv: 2405.17927."},{"key":"10.1016\/j.inffus.2026.104405_bib0122","doi-asserted-by":"crossref","first-page":"34892","DOI":"10.52202\/075280-1516","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0123","unstructured":"H. Lu, W. Liu, B. Zhang, B. Wang, K. Dong, B. Liu, J. Sun, T. Ren, Z. Li, H. Yang, et al., Deepseek-vl: towards real-world vision-language understanding, 2024, arXiv: 2403.05525."},{"key":"10.1016\/j.inffus.2026.104405_bib0124","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0125","doi-asserted-by":"crossref","unstructured":"H. Zhang, X. Li, L. Bing, Video-LLAMA: an instruction-tuned audio-visual language model for video understanding, 2023, arXiv: 2306.02858.","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"10.1016\/j.inffus.2026.104405_bib0126","unstructured":"J. Bai, S. Bai, S. Yang, S. Wang, S. Tan, P. Wang, J. Lin, C. Zhou, J. Zhou, Qwen-vl: a frontier large vision-language model with versatile abilities, 1 (2) (2023) 3. arXiv: 2308.12966."},{"key":"10.1016\/j.inffus.2026.104405_bib0127","unstructured":"Y. Jin, K. Xu, L. Chen, C. Liao, J. Tan, Q. Huang, B. Chen, C. Lei, A. Liu, C. Song, et al., Unified language-vision pretraining in LLM with dynamic discrete visual tokenization, 2023, arXiv: 2309.04669."},{"key":"10.1016\/j.inffus.2026.104405_bib0128","unstructured":"J. Zhu, X. Ding, Y. Ge, Y. Ge, S. Zhao, H. Zhao, X. Wang, Y. Shan, VL-GPT: a generative pre-trained transformer for vision and language understanding and generation, 2023a, arXiv: 2312.09251."},{"key":"10.1016\/j.inffus.2026.104405_bib0129","unstructured":"D. Zhu, J. Chen, X. Shen, X. Li, M. Elhoseiny, MiniGPT-4: enhancing vision-language understanding with advanced large language models, 2023b), arXiv: 2304.10592."},{"key":"10.1016\/j.inffus.2026.104405_bib0130","series-title":"International Conference on Machine Learning","first-page":"1931","article-title":"Unifying vision-and-language tasks via text generation","author":"Cho","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0131","unstructured":"X. Chen, J. Djolonga, P. Padlewski, B. Mustafa, S. Changpinyo, J. Wu, C.R. Ruiz, S. Goodman, X. Wang, Y. Tay, et al., Pali-x: on scaling up a multilingual vision and language model, 2023, arXiv: 2305.18565."},{"key":"10.1016\/j.inffus.2026.104405_bib0132","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: a visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0133","unstructured":"P. Gao, J. Han, R. Zhang, Z. Lin, S. Geng, A. Zhou, W. Zhang, P. Lu, C. He, X. Yue, et al., LLAMA-adapter v2: parameter-efficient visual instruction model, 2023, arXiv: 2304.15010."},{"key":"10.1016\/j.inffus.2026.104405_bib0134","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"5227","article-title":"Vl-adapter: parameter-efficient transfer learning for vision-and-language tasks","author":"Sung","year":"2022"},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0135","first-page":"3","article-title":"Lora: low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"ICLR"},{"key":"10.1016\/j.inffus.2026.104405_bib0136","series-title":"2023 IEEE International Conference on Big Data (BigData)","first-page":"2247","article-title":"Multimodal large language models: a survey","author":"Wu","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0137","unstructured":"Y. Jin, J. Li, Y. Liu, T. Gu, K. Wu, Z. Jiang, M. He, B. Zhao, X. Tan, Z. Gan, et al., Efficient multimodal large language models: a survey, 2024, arXiv: 2405.10739."},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0138","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","article-title":"Adaptive mixtures of local experts","volume":"3","author":"Jacobs","year":"1991","journal-title":"Neural comput."},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0139","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1162\/neco.1994.6.2.181","article-title":"Hierarchical mixtures of experts and the EM algorithm","volume":"6","author":"Jordan","year":"1994","journal-title":"Neural Comput."},{"key":"10.1016\/j.inffus.2026.104405_bib0140","unstructured":"D. Lepikhin, H. Lee, Y. Xu, D. Chen, O. Firat, Y. Huang, M. Krikun, N. Shazeer, Z. Chen, Gshard: scaling giant models with conditional computation and automatic sharding, 2020, arXiv: 2006.16668."},{"issue":"7","key":"10.1016\/j.inffus.2026.104405_bib0141","first-page":"3896","article-title":"A survey on mixture of experts in large language models","volume":"37","author":"Cai","year":"2025","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"10.1016\/j.inffus.2026.104405_bib0142","doi-asserted-by":"crossref","first-page":"42048","DOI":"10.52202\/079017-1330","article-title":"Mome: mixture of multimodal experts for generalist multimodal large language models","volume":"37","author":"Shen","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0143","doi-asserted-by":"crossref","first-page":"9564","DOI":"10.52202\/068431-0695","article-title":"Multimodal contrastive learning with limoe: the language-image mixture of experts","volume":"35","author":"Mustafa","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0144","series-title":"European Conference on Computer Vision","first-page":"304","article-title":"Mm1: methods, analysis and insights from multimodal llm pre-training","author":"McKinzie","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0145","unstructured":"Z. Wu, X. Chen, Z. Pan, X. Liu, W. Liu, D. Dai, H. Gao, Y. Ma, C. Wu, B. Wang, et al., Deepseek-vl2: mixture-of-experts vision-language models for advanced multimodal understanding, 2024, arXiv: 2412.10302."},{"key":"10.1016\/j.inffus.2026.104405_bib0146","unstructured":"B. Lin, Z. Tang, Y. Ye, J. Cui, B. Zhu, P. Jin, J. Huang, J. Zhang, Y. Pang, M. Ning, et al., Moe-llava: mixture of experts for large vision-language models, 2024, arXiv: 2401.15947."},{"key":"10.1016\/j.inffus.2026.104405_bib0147","doi-asserted-by":"crossref","first-page":"3424","DOI":"10.1109\/TPAMI.2025.3532688","article-title":"Uni-moe: scaling unified multimodal llms with mixture of experts","volume":"47","author":"Li","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.inffus.2026.104405_bib0148","unstructured":"X. Wu, S. Huang, F. Wei, Mixture of lora experts, 2024, arXiv: 2404.13628."},{"key":"10.1016\/j.inffus.2026.104405_bib0149","doi-asserted-by":"crossref","first-page":"27730","DOI":"10.52202\/068431-2011","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0150","doi-asserted-by":"crossref","first-page":"53728","DOI":"10.52202\/075280-2338","article-title":"Direct preference optimization: your language model is secretly a reward model","volume":"36","author":"Rafailov","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0151","unstructured":"W. Wang, Z. Chen, W. Wang, Y. Cao, Y. Liu, Z. Gao, J. Zhu, X. Zhu, L. Lu, Y. Qiao, et al., Enhancing the reasoning ability of multimodal large language models via mixed preference optimization, 2024, arXiv: 2411.10442."},{"key":"10.1016\/j.inffus.2026.104405_bib0152","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103184","article-title":"Explainability and vision foundation models: a survey","volume":"122","author":"Kazmierczak","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.inffus.2026.104405_bib0153","doi-asserted-by":"crossref","first-page":"22199","DOI":"10.52202\/068431-1613","article-title":"Large language models are zero-shot reasoners","volume":"35","author":"Kojima","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_sbref0154","article-title":"e-snli: natural language inference with natural language explanations","volume":"31","author":"Camburu","year":"2018","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0155","unstructured":"Y. Wang, S. Wu, Y. Zhang, S. Yan, Z. Liu, J. Luo, H. Fei, Multimodal chain-of-thought reasoning: a comprehensive survey, 2025, arXiv: 2503.12605."},{"key":"10.1016\/j.inffus.2026.104405_bib0156","unstructured":"Z. Chen, Q. Zhou, Y. Shen, Y. Hong, H. Zhang, C. Gan, See, think, confirm: interactive prompting between vision and language models for knowledge-based visual reasoning, 2023, arXiv: 2301.05226."},{"key":"10.1016\/j.inffus.2026.104405_bib0157","unstructured":"Z. Zhang, A. Zhang, M. Li, H. Zhao, G. Karypis, A. Smola, Multimodal chain-of-thought reasoning in language models, 2023, arXiv: 2302.00923."},{"key":"10.1016\/j.inffus.2026.104405_bib0158","unstructured":"H. Zhou, X. Li, R. Wang, M. Cheng, T. Zhou, C.-J. Hsieh, R1-zero\u2019s \u201cAha Moment\u201d in visual reasoning on a 2B non-SFT model, 2025, arXiv: 2503.05132."},{"key":"10.1016\/j.inffus.2026.104405_bib0159","unstructured":"F. Meng, L. Du, Z. Liu, Z. Zhou, Q. Lu, D. Fu, T. Han, B. Shi, W. Wang, J. He, et al., MM-EUREKA: exploring the frontiers of multimodal reasoning with rule-based reinforcement learning, 2025, arXiv: 2503.07365."},{"key":"10.1016\/j.inffus.2026.104405_bib0160","doi-asserted-by":"crossref","first-page":"11809","DOI":"10.52202\/075280-0517","article-title":"Tree of thoughts: deliberate problem solving with large language models","volume":"36","author":"Yao","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0161","unstructured":"V. Do, O.-M. Camburu, Z. Akata, T. Lukasiewicz, e-snli-ve: corrected visual-textual entailment with natural language explanations, 2020, arXiv: 2004.03744."},{"key":"10.1016\/j.inffus.2026.104405_bib0162","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"1244","article-title":"e-vil: a dataset and benchmark for natural language explanations in vision-language tasks","author":"Kayser","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0163","doi-asserted-by":"crossref","first-page":"8612","DOI":"10.52202\/079017-0275","article-title":"Visual cot: advancing multi-modal language models with a comprehensive dataset and benchmark for chain-of-thought reasoning","volume":"37","author":"Shao","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"3","key":"10.1016\/j.inffus.2026.104405_bib0164","doi-asserted-by":"crossref","DOI":"10.23915\/distill.00030","article-title":"Multimodal neurons in artificial neural networks","volume":"6","author":"Goh","year":"2021","journal-title":"Distill"},{"key":"10.1016\/j.inffus.2026.104405_bib0165","unstructured":"C. Olsson, N. Elhage, N. Nanda, N. Joseph, N. DasSarma, T. Henighan, B. Mann, A. Askell, Y. Bai, A. Chen, et al., In-context learning and induction heads, 2022, arXiv: 2209.11895."},{"key":"10.1016\/j.inffus.2026.104405_bib0166","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"11248","article-title":"Are vision-language transformers learning multimodal representations? a probing perspective","volume":"36","author":"Salin","year":"2022"},{"key":"10.1016\/j.inffus.2026.104405_bib0167","doi-asserted-by":"crossref","first-page":"17359","DOI":"10.52202\/068431-1262","article-title":"Locating and editing factual associations in GPT","volume":"35","author":"Meng","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0168","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"2856","article-title":"Towards vision-language mechanistic interpretability: a causal tracing tool for blip","author":"Palit","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0169","unstructured":"M. Golovanevsky, W. Rudman, V. Palit, R. Singh, C. Eickhoff, What do vlms notice? a mechanistic interpretability pipeline for gaussian-noise-free text-image corruption and evaluation, 2024, arXiv: 2406.16320."},{"key":"10.1016\/j.inffus.2026.104405_bib0170","unstructured":"C. Neo, L. Ong, P. Torr, M. Geva, D. Krueger, F. Barez, Towards interpreting visual information processing in vision-language models, 2024, arXiv: 2410.07149."},{"key":"10.1016\/j.inffus.2026.104405_bib0171","unstructured":"L. Gao, A. Rajaram, J. Coxon, S.V. Govande, B. Baker, D. Mossing, Weight-sparse transformers have interpretable circuits, 2025, arXiv: 2511.13653."},{"key":"10.1016\/j.inffus.2026.104405_bib0172","unstructured":"A. Nam, H. Conklin, Y. Yang, T. Griffiths, J. Cohen, S.-J. Leslie, Causal head gating: a framework for interpreting roles of attention heads in transformers, 2025, arXiv: 2505.13737."},{"key":"10.1016\/j.inffus.2026.104405_bib0173","unstructured":"Y. Gandelsman, A.A. Efros, J. Steinhardt, Interpreting the second-order effects of neurons in clip, 2024, arXiv: 2406.04341."},{"key":"10.1016\/j.inffus.2026.104405_bib0174","doi-asserted-by":"crossref","unstructured":"J. Huo, Y. Yan, B. Hu, Y. Yue, X. Hu, Mmneuron: discovering neuron-level domain-specific interpretation in multimodal large language model, 2024, arXiv: 2406.11193.","DOI":"10.18653\/v1\/2024.emnlp-main.387"},{"key":"10.1016\/j.inffus.2026.104405_bib0175","unstructured":"D. Rai, Y. Zhou, S. Feng, A. Saparov, Z. Yao, A practical review of mechanistic interpretability for transformer-based language models, 2024, arXiv: 2407.02646."},{"key":"10.1016\/j.inffus.2026.104405_bib0176","unstructured":"Z. Lin, S. Basu, M. Beigi, V. Manjunatha, R.A. Rossi, Z. Wang, Y. Zhou, S. Balasubramanian, A. Zarei, K. Rezaei, et al., A survey on mechanistic interpretability for multi-modal foundation models, 2025, arXiv: 2502.17516."},{"key":"10.1016\/j.inffus.2026.104405_bib0177","series-title":"The Fourth Blogpost Track at ICLR","article-title":"Mechanistic interpretability meets vision language models: insights and limitations","author":"Liu","year":"2025"},{"key":"10.1016\/j.inffus.2026.104405_bib0178","series-title":"International Conference on Machine Learning","first-page":"10764","article-title":"Pal: program-aided language models","author":"Gao","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0179","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing","first-page":"5153","article-title":"LINC: a neurosymbolic approach for logical reasoning by combining language models with first-order logic provers","author":"Olausson","year":"2023"},{"key":"10.1016\/j.inffus.2026.104405_bib0180","doi-asserted-by":"crossref","first-page":"45548","DOI":"10.52202\/075280-1974","article-title":"Satlm: satisfiability-aided language models using declarative prompting","volume":"36","author":"Ye","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0181","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"9590","article-title":"Visual program distillation: distilling tools and programmatic reasoning into vision-language models","author":"Hu","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0182","doi-asserted-by":"crossref","unstructured":"S. Menon, R. Zemel, C. Vondrick, Whiteboard-of-thought: thinking step-by-step across modalities, 2024, arXiv: 2406.14562.","DOI":"10.18653\/v1\/2024.emnlp-main.1117"},{"key":"10.1016\/j.inffus.2026.104405_bib0183","unstructured":"J. Wang, Z. Kang, H. Wang, H. Jiang, J. Li, B. Wu, Y. Wang, J. Ran, X. Liang, C. Feng, et al., VGR: visual grounded reasoning, 2025, arXiv: 2506.11991."},{"key":"10.1016\/j.inffus.2026.104405_bib0184","series-title":"Proceedings of the 19th International Conference on Neurosymbolic Learning and Reasoning","first-page":"420","article-title":"Towards a neurosymbolic reasoning system grounded in schematic representations","volume":"Vol. 284","author":"Olivier","year":"2025"},{"issue":"21","key":"10.1016\/j.inffus.2026.104405_bib0185","doi-asserted-by":"crossref","first-page":"12809","DOI":"10.1007\/s00521-024-09960-z","article-title":"Neuro-symbolic artificial intelligence: a survey","volume":"36","author":"Bhuyan","year":"2024","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.inffus.2026.104405_bib0186","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"21501","article-title":"Measuring cross-modal interactions in multimodal models","volume":"Vol. 39","author":"Wenderoth","year":"2025"},{"key":"10.1016\/j.inffus.2026.104405_bib0187","unstructured":"Z. Wang, K. Wang, MultiSHAP: a shapley-based framework for explaining cross-modal interactions in multimodal AI models, 2025, arXiv: 2508.00576."},{"key":"10.1016\/j.inffus.2026.104405_sbref0188","article-title":"Sanity checks for saliency maps","volume":"31","author":"Adebayo","year":"2018","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0189","unstructured":"X. Zhang, S. Li, N. Shi, B. Hauer, Z. Wu, G. Kondrak, M. Abdul-Mageed, L.V.S. Lakshmanan, Cross-modal consistency in multimodal large language models, 2024, arXiv: 2411.09273."},{"key":"10.1016\/j.inffus.2026.104405_bib0190","unstructured":"P.P. Liang, Y. Lyu, X. Fan, J. Tsaw, Y. Liu, S. Mo, D. Yogatama, L.-P. Morency, R. Salakhutdinov, High-modality multimodal transformer: quantifying modality & interaction heterogeneity for high-modality representation learning, 2022, arXiv: 2203.01311."},{"key":"10.1016\/j.inffus.2026.104405_bib0191","series-title":"Medical Imaging with Deep Learning","first-page":"978","article-title":"Multimodal image-text matching improves retrieval-based chest x-ray report generation","author":"Jeong","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0192","unstructured":"C. Agarwal, Rethinking explainability in the era of multimodal AI, 2025, arXiv: 2506.13060."},{"key":"10.1016\/j.inffus.2026.104405_sbref0193","article-title":"A benchmark for interpretability methods in deep neural networks","volume":"32","author":"Hooker","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0194","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"3429","article-title":"Interpretable explanations of black boxes by meaningful perturbation","author":"Fong","year":"2017"},{"key":"10.1016\/j.inffus.2026.104405_bib0195","unstructured":"D. Alvarez-Melis, T.S. Jaakkola, On the robustness of interpretability methods, 2018, arXiv: 1806.08049."},{"key":"10.1016\/j.inffus.2026.104405_bib0196","first-page":"841","article-title":"Counterfactual explanations without opening the black box: automated decisions and the GDPR","volume":"31","author":"Wachter","year":"2017","journal-title":"Harv. JL Tech."},{"key":"10.1016\/j.inffus.2026.104405_bib0197","doi-asserted-by":"crossref","unstructured":"A. White, A.d. Garcez, Measurable counterfactual local explanations for any classifier, 2019, arXiv: 1908.03020.","DOI":"10.3233\/FAIA200387"},{"key":"10.1016\/j.inffus.2026.104405_sbref0198","article-title":"On the (in) fidelity and sensitivity of explanations","volume":"32","author":"Yeh","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"11","key":"10.1016\/j.inffus.2026.104405_bib0199","doi-asserted-by":"crossref","first-page":"2660","DOI":"10.1109\/TNNLS.2016.2599820","article-title":"Evaluating the visualization of what a deep neural network has learned","volume":"28","author":"Samek","year":"2016","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.inffus.2026.104405_bib0200","unstructured":"V. Petsiuk, A. Das, K. Saenko, Rise: randomized input sampling for explanation of black-box models, 2018, arXiv: 1806.07421."},{"key":"10.1016\/j.inffus.2026.104405_bib0201","series-title":"Proceedings of the 27th International Conference on Artificial Intelligence and Statistics","first-page":"2071","article-title":"Sparse and faithful explanations without sparse models","volume":"238","author":"Sun","year":"2024"},{"issue":"19","key":"10.1016\/j.inffus.2026.104405_bib0202","doi-asserted-by":"crossref","first-page":"9423","DOI":"10.3390\/app12199423","article-title":"XAI systems evaluation: a review of human and computer-centred methods","volume":"12","author":"Lopes","year":"2022","journal-title":"Appl. Sci."},{"key":"10.1016\/j.inffus.2026.104405_sbref0203","doi-asserted-by":"crossref","DOI":"10.1016\/j.ijhcs.2025.103625","article-title":"User-centric evaluation of explainability of AI with and for humans: a comprehensive empirical study","volume":"205","author":"Bobek","year":"2025","journal-title":"Int. J. Hum. Comput. Stud."},{"key":"10.1016\/j.inffus.2026.104405_bib0204","doi-asserted-by":"crossref","DOI":"10.3389\/frai.2024.1456486","article-title":"Human-centered evaluation of explainable AI applications: a systematic review","volume":"7","author":"Kim","year":"2024","journal-title":"Front. Artif. Intell."},{"issue":"23","key":"10.1016\/j.inffus.2026.104405_bib0205","doi-asserted-by":"crossref","DOI":"10.3390\/app142311288","article-title":"An overview of the empirical evaluation of explainable ai (xai): a comprehensive guideline for user-centered evaluation in xai","volume":"14","author":"Naveed","year":"2024","journal-title":"Appl. Sci."},{"key":"10.1016\/j.inffus.2026.104405_bib0206","doi-asserted-by":"crossref","first-page":"101603","DOI":"10.1109\/ACCESS.2024.3431437","article-title":"Explainable artificial intelligence for autonomous driving: a comprehensive overview and field guide for future research directions","volume":"12","author":"Atakishiyev","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.inffus.2026.104405_bib0207","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"563","article-title":"Textual explanations for self-driving vehicles","author":"Kim","year":"2018"},{"issue":"12","key":"10.1016\/j.inffus.2026.104405_bib0208","doi-asserted-by":"crossref","first-page":"19342","DOI":"10.1109\/TITS.2024.3474469","article-title":"Explainable AI for safe and trustworthy autonomous driving: a systematic review","volume":"25","author":"Kuznietsov","year":"2024","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"9","key":"10.1016\/j.inffus.2026.104405_bib0209","doi-asserted-by":"crossref","first-page":"10617","DOI":"10.1109\/TITS.2024.3424667","article-title":"Grounded relational inference: domain knowledge driven explainable autonomous driving","volume":"25","author":"Tang","year":"2024","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0210","doi-asserted-by":"crossref","first-page":"867","DOI":"10.1007\/s11301-023-00320-0","article-title":"Applications of explainable artificial intelligence in finance\u2014a systematic review of finance, information systems, and computer science literature: P. Weber et al","volume":"74","author":"Weber","year":"2024","journal-title":"Manag. Rev. Q."},{"issue":"9","key":"10.1016\/j.inffus.2026.104405_bib0211","doi-asserted-by":"crossref","first-page":"4847","DOI":"10.1007\/s00521-023-09232-2","article-title":"Toward interpretable credit scoring: integrating explainable artificial intelligence with deep learning for credit card default prediction","volume":"36","author":"Talaat","year":"2024","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.inffus.2026.104405_bib0212","unstructured":"S. Serrano, N.A. Smith, Is attention interpretable?, 2019, arXiv: 1906.03731."},{"key":"10.1016\/j.inffus.2026.104405_bib0213","series-title":"Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery & Data Mining","first-page":"25","article-title":"Why attentions may not be interpretable?","author":"Bai","year":"2021"},{"key":"10.1016\/j.inffus.2026.104405_bib0214","doi-asserted-by":"crossref","unstructured":"K. Clark, U. Khandelwal, O. Levy, C.D. Manning, What does BERT look at? An analysis of BERT\u2019s attention, 2019, arXiv: 1906.04341.","DOI":"10.18653\/v1\/W19-4828"},{"key":"10.1016\/j.inffus.2026.104405_bib0215","unstructured":"S. Wiegreffe, Y. Pinter, Attention is not not explanation, 2019, arXiv: 1908.04626."},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0216","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1080\/00335558008248231","article-title":"Orienting of attention","volume":"32","author":"Posner","year":"1980","journal-title":"Q. J. Exp. Psychol."},{"issue":"6","key":"10.1016\/j.inffus.2026.104405_bib0217","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1167\/13.6.4","article-title":"Dynamic weighting of multisensory stimuli shapes decision-making in rats and humans","volume":"13","author":"Sheppard","year":"2013","journal-title":"J. Vis."},{"issue":"47","key":"10.1016\/j.inffus.2026.104405_bib0218","first-page":"1","article-title":"Efficient modality selection in multimodal learning","volume":"25","author":"He","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104405_bib0219","series-title":"The Thirty-Third International Flairs Conference","article-title":"Towards quantification of explainability in explainable artificial intelligence methods","author":"Islam","year":"2020"},{"key":"10.1016\/j.inffus.2026.104405_bib0220","unstructured":"Y. Dang, K. Huang, J. Huo, Y. Yan, S. Huang, D. Liu, M. Gao, J. Zhang, C. Qian, K. Wang, et al., Explainable and interpretable multimodal large language models: a comprehensive survey, 2024, arXiv: 2412.02104."},{"key":"10.1016\/j.inffus.2026.104405_bib0221","unstructured":"P.P. Liang, Y. Lyu, G. Chhablani, N. Jain, Z. Deng, X. Wang, L.-P. Morency, R. Salakhutdinov, Multiviz: towards visualizing and understanding multimodal models, 2022, arXiv: 2207.00056."},{"issue":"10","key":"10.1016\/j.inffus.2026.104405_bib0222","doi-asserted-by":"crossref","first-page":"e537","DOI":"10.1016\/S2589-7500(20)30218-1","article-title":"Reporting guidelines for clinical trial reports for interventions involving artificial intelligence: the CONSORT-AI extension","volume":"2","author":"Liu","year":"2020","journal-title":"Lancet Digit. Health"},{"issue":"1","key":"10.1016\/j.inffus.2026.104405_bib0223","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2022.103111","article-title":"A survey on XAI and natural language explanations","volume":"60","author":"Cambria","year":"2023","journal-title":"Inf. Process. Manag."},{"key":"10.1016\/j.inffus.2026.104405_bib0224","unstructured":"E. Cambria, L. Malandri, F. Mercorio, N. Nobani, A. Seveso, Xai meets llms: a survey of the relation between explainable ai and large language models, 2024, arXiv: 2407.15248."},{"key":"10.1016\/j.inffus.2026.104405_bib0225","unstructured":"A. Bilal, D. Ebert, B. Lin, LLMs for explainable AI: a comprehensive survey, 2025, arXiv: 2504.00125."},{"key":"10.1016\/j.inffus.2026.104405_bib0226","unstructured":"A. Palikhe, Z. Yu, Z. Wang, W. Zhang, Towards transparent AI: a survey on explainable large language models, 2025, arXiv: 2506.21812."},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0227","doi-asserted-by":"crossref","first-page":"454","DOI":"10.3758\/s13423-020-01825-5","article-title":"Artificial cognition: how experimental psychology can help generate explainable artificial intelligence","volume":"28","author":"Taylor","year":"2021","journal-title":"Psychon. Bull. Rev."},{"issue":"3","key":"10.1016\/j.inffus.2026.104405_bib0228","first-page":"5","article-title":"Artificial intelligence, autonomy, and human-machine teams\u2014interdependence, context, and explainable AI","volume":"40","author":"Lawless","year":"2019","journal-title":"Ai Mag."},{"key":"10.1016\/j.inffus.2026.104405_bib0229","series-title":"Proceedings of the 2019 CHI Conference on Human Factors in Computing Systems","first-page":"1","article-title":"Designing theory-driven user-centric explainable AI","author":"Wang","year":"2019"},{"issue":"2","key":"10.1016\/j.inffus.2026.104405_bib0230","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pbio.1002073","article-title":"Cortical hierarchies perform Bayesian causal inference in multisensory perception","volume":"13","author":"Rohe","year":"2015","journal-title":"PLoS Biol."},{"key":"10.1016\/j.inffus.2026.104405_bib0231","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"22650","article-title":"Adventures of trustworthy vision-language models: a survey","volume":"Vol. 38","author":"Vatsa","year":"2024"},{"key":"10.1016\/j.inffus.2026.104405_bib0232","doi-asserted-by":"crossref","unstructured":"J. Kunz, M. Kuhlmann, Properties and challenges of LLM-generated explanations, 2024, arXiv: 2402.10532.","DOI":"10.18653\/v1\/2024.hcinlp-1.2"}],"container-title":["Information Fusion"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526002848?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526002848?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T11:43:12Z","timestamp":1782128592000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1566253526002848"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":232,"alternative-id":["S1566253526002848"],"URL":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104405","relation":{},"ISSN":["1566-2535"],"issn-type":[{"value":"1566-2535","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Decoding the multimodal maze: A systematic review on the adoption of explainability in multimodal attention-based models","name":"articletitle","label":"Article Title"},{"value":"Information Fusion","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104405","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"104405"}}