{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T14:02:14Z","timestamp":1785074534219,"version":"3.55.0"},"reference-count":49,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116392","type":"journal-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T23:35:00Z","timestamp":1781566500000},"page":"116392","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Graph Mixture of Experts with Differential Cross-Attention Alignment for Multimodal Intent Recognition"],"prefix":"10.1016","volume":"349","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-8897-3737","authenticated-orcid":false,"given":"Shilin","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenbin","family":"An","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0751-2602","authenticated-orcid":false,"given":"Qidong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fang","family":"Nan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiahao","family":"Nie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8864-8412","authenticated-orcid":false,"given":"Zhi","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xian-Sheng","family":"Hua","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaqiang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7888-0587","authenticated-orcid":false,"given":"Feng","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116392_b1","series-title":"17th Annual Conference of the International Speech Communication Association, Interspeech 2016, San Francisco, CA, USA, September 8-12, 2016","first-page":"685","article-title":"Attention-based recurrent neural network models for joint intent detection and slot filling","author":"Liu","year":"2016"},{"key":"10.1016\/j.knosys.2026.116392_b2","series-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing","first-page":"2078","article-title":"A stack-propagation framework with token-level intent detection for spoken language understanding","author":"Qin","year":"2019"},{"issue":"4","key":"10.1016\/j.knosys.2026.116392_b3","doi-asserted-by":"crossref","first-page":"21","DOI":"10.1109\/MIS.2023.3283909","article-title":"New user intent discovery with robust pseudo label training and source domain joint training","volume":"38","author":"An","year":"2023","journal-title":"IEEE Intell. Syst."},{"key":"10.1016\/j.knosys.2026.116392_b4","first-page":"15365","article-title":"Unleashing the potential of model bias for generalized category discovery","volume":"vol. 39","author":"An","year":"2025"},{"key":"10.1016\/j.knosys.2026.116392_b5","doi-asserted-by":"crossref","unstructured":"H. Zhang, H. Xu, X. Wang, Q. Zhou, S. Zhao, J. Teng, Mintrec: A new dataset for multimodal intent recognition, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 1688\u20131697.","DOI":"10.1145\/3503161.3547906"},{"key":"10.1016\/j.knosys.2026.116392_b6","series-title":"A review of multimodal explainable artificial intelligence: Past, present and future","author":"Sun","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b7","first-page":"6558","article-title":"Multimodal transformer for unaligned multimodal language sequences","volume":"vol. 2019","author":"Tsai","year":"2019"},{"key":"10.1016\/j.knosys.2026.116392_b8","doi-asserted-by":"crossref","unstructured":"J. Wu, S. Mai, H. Hu, Graph capsule aggregation for unaligned multimodal sequences, in: Proceedings of the 2021 International Conference on Multimodal Interaction, 2021, pp. 521\u2013529.","DOI":"10.1145\/3462244.3479931"},{"key":"10.1016\/j.knosys.2026.116392_b9","first-page":"17114","article-title":"Token-level contrastive learning with modality-aware prompting for multimodal intent recognition","volume":"vol. 38","author":"Zhou","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b10","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"10206","article-title":"Sdif-da: A shallow-to-deep interaction framework with data augmentation for multi-modal intent detection","author":"Huang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b11","doi-asserted-by":"crossref","unstructured":"D. Hazarika, R. Zimmermann, S. Poria, Misa: Modality-invariant and-specific representations for multimodal sentiment analysis, in: Proceedings of the 28th ACM International Conference on Multimedia, 2020, pp. 1122\u20131131.","DOI":"10.1145\/3394171.3413678"},{"key":"10.1016\/j.knosys.2026.116392_b12","first-page":"2359","article-title":"Integrating multimodal information in large pretrained transformers","volume":"vol. 2020","author":"Rahman","year":"2020"},{"key":"10.1016\/j.knosys.2026.116392_b13","series-title":"2024 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Multi-modal intent detection with lvamoe: the language-visual-audio mixture of experts","author":"Li","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b14","doi-asserted-by":"crossref","unstructured":"W. An, F. Tian, S. Leng, J. Nie, H. Lin, Q. Wang, P. Chen, X. Zhang, S. Lu, Mitigating object hallucinations in large vision-language models with assembly of global and local attention, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 29915\u201329926.","DOI":"10.1109\/CVPR52734.2025.02784"},{"key":"10.1016\/j.knosys.2026.116392_b15","series-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","article-title":"Differential transformer","author":"Ye","year":"2025"},{"key":"10.1016\/j.knosys.2026.116392_b16","first-page":"17267","article-title":"Adaptive multimodal fusion: Dynamic attention allocation for intent recognition","volume":"vol. 39","author":"Hu","year":"2025"},{"key":"10.1016\/j.knosys.2026.116392_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109537","article-title":"Line graph contrastive learning for link prediction","volume":"140","author":"Zhang","year":"2023","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116392_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112056","article-title":"BP-MoE: Behavior pattern-aware mixture-of-experts for temporal graph representation learning","volume":"299","author":"Chen","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116392_b19","doi-asserted-by":"crossref","unstructured":"Y. Fang, W. Huang, G. Wan, K. Su, M. Ye, EMOE: Modality-Specific Enhanced Dynamic Emotion Experts, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 14314\u201314324.","DOI":"10.1109\/CVPR52734.2025.01335"},{"key":"10.1016\/j.knosys.2026.116392_b20","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113524","article-title":"Knowledge-augmented encoder for few-shot deep intent recognition in air traffic control","volume":"320","author":"Hui","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116392_b21","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113541","article-title":"MESN: A multimodal knowledge graph embedding framework with expert fusion and relational attention","volume":"318","author":"Tran","year":"2025","journal-title":"Knowl.-Based Syst."},{"issue":"12","key":"10.1016\/j.knosys.2026.116392_b22","doi-asserted-by":"crossref","first-page":"24","DOI":"10.1016\/j.eng.2025.01.012","article-title":"Machine memory intelligence: Inspired by human memory mechanisms","volume":"55","author":"Zheng","year":"2025","journal-title":"Engineering"},{"key":"10.1016\/j.knosys.2026.116392_b23","unstructured":"H. Zhang, X. Wang, H. Xu, Q. Zhou, K. Gao, J. Su, J. Zhao, W. Li, Y. Chen, MIntRec2.0: A Large-scale Benchmark Dataset for Multimodal Intent Recognition and Out-of-scope Detection in Conversations, in: The Twelfth International Conference on Learning Representations, 2024."},{"key":"10.1016\/j.knosys.2026.116392_b24","doi-asserted-by":"crossref","unstructured":"T. Saha, A. Patra, S. Saha, P. Bhattacharyya, Towards emotion-aided multi-modal dialogue act classification, in: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, 2020, pp. 4361\u20134372.","DOI":"10.18653\/v1\/2020.acl-main.402"},{"key":"10.1016\/j.knosys.2026.116392_b25","doi-asserted-by":"crossref","unstructured":"A. Graves, S. Fern\u00e1ndez, F. Gomez, J. Schmidhuber, Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks, in: Proceedings of the 23rd International Conference on Machine Learning, 2006, pp. 369\u2013376.","DOI":"10.1145\/1143844.1143891"},{"key":"10.1016\/j.knosys.2026.116392_b26","doi-asserted-by":"crossref","unstructured":"K. Sun, Z. Xie, M. Ye, H. Zhang, Contextual augmented global contrast for multimodal intent recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26963\u201326973.","DOI":"10.1109\/CVPR52733.2024.02546"},{"key":"10.1016\/j.knosys.2026.116392_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.126373","article-title":"An effective multimodal representation and fusion method for multimodal intent recognition","volume":"548","author":"Huang","year":"2023","journal-title":"Neurocomputing"},{"key":"10.1016\/j.knosys.2026.116392_b28","doi-asserted-by":"crossref","unstructured":"Z. Zhu, X. Cheng, Z. Chen, Y. Chen, Y. Zhang, X. Wu, Y. Zheng, B. Xing, InMu-Net: advancing multi-modal intent detection via information bottleneck and multi-sensory processing, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 515\u2013524.","DOI":"10.1145\/3664647.3681623"},{"key":"10.1016\/j.knosys.2026.116392_b29","unstructured":"Q. Yang, X. Li, F. Lin, M. Ye, Adaptive Re-calibration Learning for Balanced Multimodal Intention Recognition, in: The Thirty-Ninth Annual Conference on Neural Information Processing Systems."},{"key":"10.1016\/j.knosys.2026.116392_b30","series-title":"European Conference on Computer Vision","first-page":"144","article-title":"Synergy of sight and semantics: visual intention understanding with clip","author":"Yang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b31","doi-asserted-by":"crossref","unstructured":"Q. Yang, Q. Shi, T. Wang, M. Ye, Uncertain multimodal intention and emotion understanding in the wild, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 24700\u201324709.","DOI":"10.1109\/CVPR52734.2025.02300"},{"key":"10.1016\/j.knosys.2026.116392_b32","series-title":"Emollm: Multimodal emotional understanding meets large language models","author":"Yang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b33","doi-asserted-by":"crossref","first-page":"110805","DOI":"10.52202\/079017-3518","article-title":"Emotion-llama: Multimodal emotion recognition and reasoning with instruction tuning","volume":"37","author":"Cheng","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1","key":"10.1016\/j.knosys.2026.116392_b34","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","article-title":"Adaptive mixtures of local experts","volume":"3","author":"Jacobs","year":"1991","journal-title":"Neural Comput."},{"key":"10.1016\/j.knosys.2026.116392_b35","series-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","author":"Shazeer","year":"2017"},{"key":"10.1016\/j.knosys.2026.116392_b36","series-title":"Gshard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2020"},{"key":"10.1016\/j.knosys.2026.116392_b37","doi-asserted-by":"crossref","unstructured":"Q. Liu, X. Wu, X. Zhao, Y. Zhu, D. Xu, F. Tian, Y. Zheng, When moe meets llms: Parameter efficient fine-tuning for multi-task medical applications, in: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, 2024, pp. 1104\u20131114.","DOI":"10.1145\/3626772.3657722"},{"key":"10.1016\/j.knosys.2026.116392_b38","doi-asserted-by":"crossref","first-page":"9564","DOI":"10.52202\/068431-0695","article-title":"Multimodal contrastive learning with limoe: the language-image mixture of experts","volume":"35","author":"Mustafa","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116392_b39","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"10006","article-title":"MMoE: Enhancing multimodal models with mixtures of multimodal interaction experts","author":"Yu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116392_b40","doi-asserted-by":"crossref","unstructured":"W. Xu, H. Jiang, X. Liang, Leveraging Knowledge of Modality Experts for Incomplete Multimodal Learning, in: Proceedings of the 32nd ACM International Conference on Multimedia, 2024, pp. 438\u2013446.","DOI":"10.1145\/3664647.3681683"},{"key":"10.1016\/j.knosys.2026.116392_b41","doi-asserted-by":"crossref","unstructured":"B. Zhou, Y. Zhang, Y. Zhao, X. Sui, X. Yuan, Multimodal graph-based variational mixture of experts network for zero-shot multimodal information extraction, in: Proceedings of the ACM on Web Conference 2025, 2025, pp. 4823\u20134831.","DOI":"10.1145\/3696410.3714832"},{"key":"10.1016\/j.knosys.2026.116392_b42","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"A MoE multimodal graph attention network framework for multimodal emotion recognition","author":"Zhang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116392_b43","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.knosys.2026.116392_b44","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116392_b45","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, B. Guo, Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.knosys.2026.116392_b46","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.127063","article-title":"Roformer: Enhanced transformer with rotary position embedding","volume":"568","author":"Su","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.knosys.2026.116392_b47","article-title":"Root mean square layer normalization","volume":"32","author":"Zhang","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116392_b48","unstructured":"T.N. Kipf, M. Welling, Semi-Supervised Classification with Graph Convolutional Networks, in: International Conference on Learning Representations, 2017."},{"key":"10.1016\/j.knosys.2026.116392_b49","series-title":"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","first-page":"1009","article-title":"MTAG: Modal-temporal attention graph for unaligned human multimodal language sequences","author":"Yang","year":"2021"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126011184?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126011184?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T13:31:34Z","timestamp":1785072694000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126011184"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":49,"alternative-id":["S0950705126011184"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116392","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Graph Mixture of Experts with Differential Cross-Attention Alignment for Multimodal Intent Recognition","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116392","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116392"}}