{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T07:20:54Z","timestamp":1783063254659,"version":"3.54.6"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114306","type":"journal-article","created":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:14:55Z","timestamp":1782198895000},"page":"114306","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["Causal mixture-of-experts for robust multimodal fusion"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2741-2174","authenticated-orcid":false,"given":"Yang","family":"Lan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9255-3268","authenticated-orcid":false,"given":"Xin","family":"Long","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaoyuan","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huiling","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jungang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"2","key":"10.1016\/j.patcog.2026.114306_b1","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1109\/TPAMI.2018.2798607","article-title":"Multimodal machine learning: A survey and taxonomy","volume":"41","author":"Baltru\u0161aitis","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"6","key":"10.1016\/j.patcog.2026.114306_b2","doi-asserted-by":"crossref","first-page":"4940","DOI":"10.1109\/TPAMI.2025.3546356","article-title":"Reliable representation learning for incomplete multi-view missing multi-label classification","volume":"47","author":"Liu","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114306_b3","doi-asserted-by":"crossref","first-page":"7635","DOI":"10.1109\/TNNLS.2022.3145048","article-title":"Adversarial multiview clustering networks with adaptive fusion","volume":"34","author":"Wang","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.114306_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.102970","article-title":"Hallucinations of large multimodal models: Problem and countermeasures","volume":"118","author":"Sun","year":"2025","journal-title":"Inf. Fusion"},{"issue":"1","key":"10.1016\/j.patcog.2026.114306_b5","doi-asserted-by":"crossref","first-page":"236","DOI":"10.1109\/TPAMI.2025.3603677","article-title":"Partial multiview incomplete multilabel learning via uncertainty-driven reliable dynamic fusion","volume":"48","author":"Wen","year":"2026","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114306_b6","article-title":"Learning compact semantic information and reliable pseudo-labels for incomplete multi-view multi-label classification","author":"Liu","year":"2026","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114306_b7","article-title":"Dual self-supervised deep graph clustering","author":"Wang","year":"2026","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.114306_b8","doi-asserted-by":"crossref","unstructured":"Y.-H.H. Tsai, S. Bai, P.P. Liang, J.Z. Kolter, L.-P. Morency, R. Salakhutdinov, Multimodal transformer for unaligned multimodal language sequences, in: Proceedings of the Conference. Association for Computational Linguistics. Meeting, Vol. 2019, 2019, p. 6558.","DOI":"10.18653\/v1\/P19-1656"},{"key":"10.1016\/j.patcog.2026.114306_b9","doi-asserted-by":"crossref","unstructured":"A. Zadeh, M. Chen, S. Poria, E. Cambria, L.-P. Morency, Tensor fusion network for multimodal sentiment analysis, in: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing, 2017, pp. 1103\u20131114.","DOI":"10.18653\/v1\/D17-1115"},{"key":"10.1016\/j.patcog.2026.114306_b10","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Shen, V.B. Lakshminarasimhan, P.P. Liang, A.B. Zadeh, L.-P. Morency, Efficient low-rank multimodal fusion with modality-specific factors, in: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2018, pp. 2247\u20132256.","DOI":"10.18653\/v1\/P18-1209"},{"key":"10.1016\/j.patcog.2026.114306_b11","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"MoE++: Accelerating mixture-of-experts methods with zero-computation experts","author":"Jin","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b12","doi-asserted-by":"crossref","unstructured":"Y. Fang, W. Huang, G. Wan, K. Su, M. Ye, EMOE: Modality-Specific Enhanced Dynamic Emotion Experts, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 14314\u201314324.","DOI":"10.1109\/CVPR52734.2025.01335"},{"issue":"1","key":"10.1016\/j.patcog.2026.114306_b13","doi-asserted-by":"crossref","DOI":"10.3390\/math12010085","article-title":"An out-of-distribution generalization framework based on variational backdoor adjustment","volume":"12","author":"Su","year":"2024","journal-title":"Mathematics"},{"issue":"5","key":"10.1016\/j.patcog.2026.114306_b14","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3644392","article-title":"Enhancing out-of-distribution generalization on graphs via causal attention learning","volume":"18","author":"Sui","year":"2024","journal-title":"ACM Trans. Knowl. Discov. from Data"},{"issue":"none","key":"10.1016\/j.patcog.2026.114306_b15","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1214\/09-SS057","article-title":"Causal inference in statistics: An overview","volume":"3","author":"Pearl","year":"2009","journal-title":"Stat. Surv."},{"key":"10.1016\/j.patcog.2026.114306_b16","unstructured":"T. Makino, K.J. Geras, K. Cho, Mitigating input-causing confounding in multimodal learning via the backdoor adjustment, in: NeurIPS 2022 Workshop on Causality for Real-World Impact, 2022."},{"key":"10.1016\/j.patcog.2026.114306_b17","series-title":"International Conference on Machine Learning","first-page":"65738","article-title":"Towards the causal complete cause of multi-modal representation learning","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b18","doi-asserted-by":"crossref","first-page":"2493","DOI":"10.1109\/TMM.2020.3013408","article-title":"Adaptive graph completion based incomplete multi-view clustering","volume":"23","author":"Wen","year":"2020","journal-title":"IEEE Trans. Multimed."},{"issue":"8","key":"10.1016\/j.patcog.2026.114306_b19","doi-asserted-by":"crossref","first-page":"10539","DOI":"10.1109\/TNNLS.2023.3242473","article-title":"Projective incomplete multi-view clustering","volume":"35","author":"Deng","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.patcog.2026.114306_b20","doi-asserted-by":"crossref","unstructured":"A. Zadeh, P.P. Liang, S. Poria, P. Vij, E. Cambria, L.-P. Morency, Multi-attention recurrent network for human communication comprehension, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 32, 2018.","DOI":"10.1609\/aaai.v32i1.12024"},{"key":"10.1016\/j.patcog.2026.114306_b21","doi-asserted-by":"crossref","unstructured":"A. Zadeh, P.P. Liang, N. Mazumder, S. Poria, E. Cambria, L.-P. Morency, Memory fusion network for multi-view sequential learning, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 32, 2018.","DOI":"10.1609\/aaai.v32i1.12021"},{"key":"10.1016\/j.patcog.2026.114306_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110847","article-title":"Sentiment analysis based on text information enhancement and multimodal feature fusion","volume":"156","author":"Liu","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114306_b23","doi-asserted-by":"crossref","first-page":"333","DOI":"10.1109\/TMM.2025.3623514","article-title":"Self-guided discriminative locality preserving projections","volume":"28","author":"Wang","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.114306_b24","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Intrinsic user-centric interpretability through global mixture of experts","author":"Swamy","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b25","series-title":"International Conference on Machine Learning","first-page":"63169","article-title":"Cooperation of experts: Fusing heterogeneous information with large margin","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b26","series-title":"GMoPE: A prompt-expert mixture framework for graph foundation models","author":"Wang","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b27","series-title":"Proceedings of the 42nd International Conference on Machine Learning","first-page":"68870","article-title":"I2moe: Interpretable multimodal interaction-aware mixture-of-experts","volume":"vol. 267","author":"Xin","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b28","series-title":"Causal debiasing medical multimodal representation learning with missing modalities","author":"Zhu","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b29","doi-asserted-by":"crossref","unstructured":"B. Liu, D. Wang, X. Yang, Y. Zhou, R. Yao, Z. Shao, J. Zhao, Show, deconfound and tell: Image captioning with causal inference, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 18041\u201318050.","DOI":"10.1109\/CVPR52688.2022.01751"},{"key":"10.1016\/j.patcog.2026.114306_b30","doi-asserted-by":"crossref","unstructured":"J. Qi, Y. Niu, J. Huang, H. Zhang, Two causal principles for improving visual dialog, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 10860\u201310869.","DOI":"10.1109\/CVPR42600.2020.01087"},{"key":"10.1016\/j.patcog.2026.114306_b31","series-title":"A causality-aware spatiotemporal model for multi-region and multi-pollutant air quality forecasting","author":"Lu","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b32","series-title":"High dimensional causal inference with variational backdoor adjustment","author":"Israel","year":"2024"},{"key":"10.1016\/j.patcog.2026.114306_b33","series-title":"Seeking the sufficiency and necessity causal features in multimodal representation learning","author":"Chen","year":"2024"},{"key":"10.1016\/j.patcog.2026.114306_b34","series-title":"Findings of the Association for Computational Linguistics: ACL 2025","first-page":"5509","article-title":"Multimodal causal reasoning benchmark: Challenging multimodal large language models to discern causal links across modalities","author":"Li","year":"2025"},{"key":"10.1016\/j.patcog.2026.114306_b35","series-title":"2015 IEEE International Conference on Image Processing","first-page":"168","article-title":"UTD-MHAD: A multimodal dataset for human action recognition utilizing a depth camera and a wearable inertial sensor","author":"Chen","year":"2015"},{"key":"10.1016\/j.patcog.2026.114306_b36","doi-asserted-by":"crossref","unstructured":"Q. Kong, Z. Wu, Z. Deng, M. Klinkigt, B. Tong, T. Murakami, Mmact: A large-scale dataset for cross modal human action understanding, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 8658\u20138667.","DOI":"10.1109\/ICCV.2019.00875"},{"key":"10.1016\/j.patcog.2026.114306_b37","series-title":"Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos","author":"Zadeh","year":"2016"},{"key":"10.1016\/j.patcog.2026.114306_b38","doi-asserted-by":"crossref","unstructured":"A.B. Zadeh, P.P. Liang, S. Poria, E. Cambria, L.-P. Morency, Multimodal language analysis in the wild: Cmu-mosei dataset and interpretable dynamic fusion graph, in: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2018, pp. 2236\u20132246.","DOI":"10.18653\/v1\/P18-1208"},{"key":"10.1016\/j.patcog.2026.114306_b39","series-title":"Grand Challenge and Workshop on Human Multimodal Language","first-page":"11","article-title":"Recognizing emotions in video using multimodal DNN feature fusion","author":"Williams","year":"2018"},{"key":"10.1016\/j.patcog.2026.114306_b40","doi-asserted-by":"crossref","unstructured":"H. Pham, P.P. Liang, T. Manzini, L.-P. Morency, B. P\u00f3czos, Found in translation: Learning robust joint representations by cyclic translations between modalities, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 33, 2019, pp. 6892\u20136899.","DOI":"10.1609\/aaai.v33i01.33016892"},{"key":"10.1016\/j.patcog.2026.114306_b41","series-title":"International Conference on Learning Representations","article-title":"Learning factorized multimodal representations","author":"Tsai","year":"2019"},{"key":"10.1016\/j.patcog.2026.114306_b42","doi-asserted-by":"crossref","unstructured":"D. Hazarika, R. Zimmermann, S. Poria, Misa: Modality-invariant and-specific representations for multimodal sentiment analysis, in: Proceedings of the 28th ACM International Conference on Multimedia, 2020, pp. 1122\u20131131.","DOI":"10.1145\/3394171.3413678"},{"key":"10.1016\/j.patcog.2026.114306_b43","doi-asserted-by":"crossref","unstructured":"W. Yu, H. Xu, Z. Yuan, J. Wu, Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 35, 2021, pp. 10790\u201310797.","DOI":"10.1609\/aaai.v35i12.17289"},{"key":"10.1016\/j.patcog.2026.114306_b44","doi-asserted-by":"crossref","unstructured":"W. Han, H. Chen, S. Poria, Improving multimodal fusion with hierarchical mutual information maximization for multimodal sentiment analysis, in: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, 2021, pp. 9180\u20139192.","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"key":"10.1016\/j.patcog.2026.114306_b45","doi-asserted-by":"crossref","unstructured":"Y. Li, Y. Wang, Z. Cui, Decoupled multimodal distilling for emotion recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 6631\u20136640.","DOI":"10.1109\/CVPR52729.2023.00641"},{"key":"10.1016\/j.patcog.2026.114306_b46","doi-asserted-by":"crossref","unstructured":"J. Devlin, M.-W. Chang, K. Lee, K. Toutanova, Bert: Pre-training of deep bidirectional transformers for language understanding, in: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), 2019, pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.patcog.2026.114306_b47","series-title":"2016 IEEE Winter Conference on Applications of Computer Vision","first-page":"1","article-title":"Openface: an open source facial behavior analysis toolkit","author":"Baltru\u0161aitis","year":"2016"},{"key":"10.1016\/j.patcog.2026.114306_b48","first-page":"1","article-title":"Multimodal feature fusion network with text difference enhancement for remote sensing change detection","volume":"63","author":"Zhou","year":"2025","journal-title":"IEEE Trans. Geosci. Remote Sens."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012719?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326012719?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T07:10:04Z","timestamp":1783062604000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326012719"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":48,"alternative-id":["S0031320326012719"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114306","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Causal mixture-of-experts for robust multimodal fusion","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114306","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114306"}}