{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:38Z","timestamp":1779228398208,"version":"3.51.4"},"reference-count":59,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100005144","name":"Qualcomm","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100005144","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004541","name":"Ministry of Education","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004541","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101965","type":"journal-article","created":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T08:50:59Z","timestamp":1772095859000},"page":"101965","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A Mixture-of-Experts model for multimodal emotion recognition in conversations"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-8857-4923","authenticated-orcid":false,"given":"Soumya","family":"Dutta","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Smruthi","family":"Balaji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sriram","family":"Ganapathy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101965_b1","article-title":"Der-gcn: Dialog and event relation-aware graph convolutional neural network for multimodal dialog emotion recognition","author":"Ai","year":"2024","journal-title":"IEEE Trans. Neural Networks Learn. Syst."},{"key":"10.1016\/j.csl.2026.101965_b2","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101965_b3","unstructured":"BehnamGhader, P., Adlakha, V., Mosbach, M., Bahdanau, D., Chapados, N., Reddy, S., 2024. LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders. In: First Conference on Language Modeling."},{"key":"10.1016\/j.csl.2026.101965_b4","doi-asserted-by":"crossref","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","article-title":"IEMOCAP: Interactive emotional dyadic motion capture database","volume":"42","author":"Busso","year":"2008","journal-title":"Lang. Resour. Eval."},{"key":"10.1016\/j.csl.2026.101965_b5","doi-asserted-by":"crossref","unstructured":"Chudasama, V., Kar, P., Gudmalwar, A., Shah, N., Wasnik, P., Onoe, N., 2022. M2fnet: Multi-modal fusion network for emotion recognition in conversation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4652\u20134661.","DOI":"10.1109\/CVPRW56347.2022.00511"},{"key":"10.1016\/j.csl.2026.101965_b6","series-title":"2014 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"960","article-title":"COVAREP\u2014A collaborative voice analysis repository for speech technologies","author":"Degottex","year":"2014"},{"key":"10.1016\/j.csl.2026.101965_b7","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K., 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.csl.2026.101965_b8","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6917","article-title":"Multimodal transformer with learnable frontend and self attention for emotion recognition","author":"Dutta","year":"2022"},{"key":"10.1016\/j.csl.2026.101965_b9","series-title":"HCAM\u2013hierarchical cross attention model for multi-modal emotion recognition","author":"Dutta","year":"2023"},{"key":"10.1016\/j.csl.2026.101965_b10","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Llm supervised pre-training for multimodal emotion recognition in conversations","author":"Dutta","year":"2025"},{"key":"10.1016\/j.csl.2026.101965_b11","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., Schuller, B., 2010. Opensmile: the munich versatile and fast open-source audio feature extractor. In: Proceedings of the 18th ACM International Conference on Multimedia. pp. 1459\u20131462.","DOI":"10.1145\/1873951.1874246"},{"key":"10.1016\/j.csl.2026.101965_b12","series-title":"Ckerc: Joint large language models with commonsense knowledge for emotion recognition in conversation","author":"Fu","year":"2024"},{"key":"10.1016\/j.csl.2026.101965_b13","series-title":"Emotion detection and analysis on social media","author":"Gaind","year":"2019"},{"key":"10.1016\/j.csl.2026.101965_b14","series-title":"2019 11th International Conference on Communication Systems & Networks","first-page":"496","article-title":"EmoKey: An emotion-aware smartphone keyboard for mental health monitoring","author":"Ghosh","year":"2019"},{"key":"10.1016\/j.csl.2026.101965_b15","doi-asserted-by":"crossref","unstructured":"Han, W., Chen, H., Poria, S., 2021. Improving Multimodal Fusion with Hierarchical Mutual Information Maximization for Multimodal Sentiment Analysis. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing. pp. 9180\u20139192.","DOI":"10.18653\/v1\/2021.emnlp-main.723"},{"key":"10.1016\/j.csl.2026.101965_b16","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Poria, S., Zadeh, A., Cambria, E., Morency, L.-P., Zimmermann, R., 2018. Conversational Memory Network for Emotion Recognition in Dyadic Dialogue Videos. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers). pp. 2122\u20132132.","DOI":"10.18653\/v1\/N18-1193"},{"key":"10.1016\/j.csl.2026.101965_b17","doi-asserted-by":"crossref","unstructured":"Hazarika, D., Zimmermann, R., Poria, S., 2020. Misa: Modality-invariant and-specific representations for multimodal sentiment analysis. In: Proceedings of the 28th ACM International Conference on Multimedia. pp. 1122\u20131131.","DOI":"10.1145\/3394171.3413678"},{"key":"10.1016\/j.csl.2026.101965_b18","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"10.1016\/j.csl.2026.101965_b19","doi-asserted-by":"crossref","unstructured":"Hu, D., Bao, Y., Wei, L., Zhou, W., Hu, S., 2023. Supervised Adversarial Contrastive Learning for Emotion Recognition in Conversations. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 10835\u201310852.","DOI":"10.18653\/v1\/2023.acl-long.606"},{"key":"10.1016\/j.csl.2026.101965_b20","doi-asserted-by":"crossref","unstructured":"Hu, G., Lin, T.-E., Zhao, Y., Lu, G., Wu, Y., Li, Y., 2022. UniMSE: Towards Unified Multimodal Sentiment Analysis and Emotion Recognition. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. pp. 7837\u20137851.","DOI":"10.18653\/v1\/2022.emnlp-main.534"},{"key":"10.1016\/j.csl.2026.101965_b21","doi-asserted-by":"crossref","unstructured":"Hu, J., Liu, Y., Zhao, J., Jin, Q., 2021. MMGCN: Multimodal Fusion via Deep Graph Convolution Network for Emotion Recognition in Conversation. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). pp. 5666\u20135675.","DOI":"10.18653\/v1\/2021.acl-long.440"},{"issue":"2","key":"10.1016\/j.csl.2026.101965_b22","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"ICLR"},{"key":"10.1016\/j.csl.2026.101965_b23","first-page":"18661","article-title":"Supervised contrastive learning","volume":"33","author":"Khosla","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101965_b24","series-title":"InstructERC: Reforming emotion recognition in conversation with multi-task retrieval-augmented large language models","author":"Lei","year":"2023"},{"key":"10.1016\/j.csl.2026.101965_b25","unstructured":"Lepikhin, D., Lee, H., Xu, Y., Chen, D., Firat, O., Huang, Y., Krikun, M., Shazeer, N., Chen, Z., 2021. GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding. In: International Conference on Learning Representations."},{"key":"10.1016\/j.csl.2026.101965_b26","doi-asserted-by":"crossref","unstructured":"Li, B., Fei, H., Liao, L., Zhao, Y., Teng, C., Chua, T.-S., Ji, D., Li, F., 2023. Revisiting disentanglement and fusion on modality and context in conversational multimodal emotion recognition. In: Proceedings of the 31st ACM International Conference on Multimedia. pp. 5923\u20135934.","DOI":"10.1145\/3581783.3612053"},{"key":"10.1016\/j.csl.2026.101965_b27","series-title":"International Conference on Machine Learning","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.csl.2026.101965_b28","article-title":"CFN-ESA: A cross-modal fusion network with emotion-shift awareness for dialogue emotion recognition","author":"Li","year":"2024","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.csl.2026.101965_b29","doi-asserted-by":"crossref","unstructured":"Li, S., Yan, H., Qiu, X., 2022. Contrast and generation make bart a good dialogue emotion recognizer. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 36, pp. 11002\u201311010.","DOI":"10.1609\/aaai.v36i10.21348"},{"issue":"3","key":"10.1016\/j.csl.2026.101965_b30","doi-asserted-by":"crossref","first-page":"2415","DOI":"10.1109\/TAFFC.2022.3141237","article-title":"Smin: Semi-supervised multi-modal interaction network for conversational emotion recognition","volume":"14","author":"Lian","year":"2022","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.csl.2026.101965_b31","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P., 2017. Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision. pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"10.1016\/j.csl.2026.101965_b32","series-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.csl.2026.101965_b33","doi-asserted-by":"crossref","unstructured":"Mai, S., Hu, H., Xing, S., 2019. Divide, conquer and combine: Hierarchical feature fusion network with local and global perspectives for multimodal affective computing. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. pp. 481\u2013492.","DOI":"10.18653\/v1\/P19-1046"},{"key":"10.1016\/j.csl.2026.101965_b34","doi-asserted-by":"crossref","unstructured":"Majumder, N., Poria, S., Hazarika, D., Mihalcea, R., Gelbukh, A., Cambria, E., 2019. Dialoguernn: An attentive rnn for emotion detection in conversations. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 33, pp. 6818\u20136825.","DOI":"10.1609\/aaai.v33i01.33016818"},{"key":"10.1016\/j.csl.2026.101965_b35","article-title":"Distributed representations of words and phrases and their compositionality","volume":"26","author":"Mikolov","year":"2013","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101965_b36","doi-asserted-by":"crossref","unstructured":"Pantic, M., Sebe, N., Cohn, J.F., Huang, T., 2005. Affective multimodal human-computer interaction. In: Proceedings of the 13th Annual ACM International Conference on Multimedia. pp. 669\u2013676.","DOI":"10.1145\/1101149.1101299"},{"key":"10.1016\/j.csl.2026.101965_b37","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D., 2014. Glove: Global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing. EMNLP, pp. 1532\u20131543.","DOI":"10.3115\/v1\/D14-1162"},{"key":"10.1016\/j.csl.2026.101965_b38","doi-asserted-by":"crossref","unstructured":"Poria, S., Cambria, E., Gelbukh, A., 2015. Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis. In: Proceedings of the 2015 Conference on Empirical Methods in Natural Language Processing. pp. 2539\u20132544.","DOI":"10.18653\/v1\/D15-1303"},{"key":"10.1016\/j.csl.2026.101965_b39","doi-asserted-by":"crossref","unstructured":"Poria, S., Hazarika, D., Majumder, N., Naik, G., Cambria, E., Mihalcea, R., 2019a. MELD: A Multimodal Multi-Party Dataset for Emotion Recognition in Conversations. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics. pp. 527\u2013536.","DOI":"10.18653\/v1\/P19-1050"},{"key":"10.1016\/j.csl.2026.101965_b40","doi-asserted-by":"crossref","first-page":"100943","DOI":"10.1109\/ACCESS.2019.2929050","article-title":"Emotion recognition in conversation: Research challenges, datasets, and recent advances","volume":"7","author":"Poria","year":"2019","journal-title":"IEEE Access"},{"key":"10.1016\/j.csl.2026.101965_b41","first-page":"8583","article-title":"Scaling vision with sparse mixture of experts","volume":"34","author":"Riquelme","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101965_b42","unstructured":"Shazeer, N., Mirhoseini, A., Maziarz, K., Davis, A., Le, Q., Hinton, G., Dean, J., 2017. Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer. In: International Conference on Learning Representations."},{"key":"10.1016\/j.csl.2026.101965_b43","doi-asserted-by":"crossref","DOI":"10.1109\/TAFFC.2025.3558222","article-title":"Towards speaker-unknown emotion recognition in conversation via progressive contrastive deep supervision","author":"Shen","year":"2025","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.csl.2026.101965_b44","doi-asserted-by":"crossref","unstructured":"Shi, T., Huang, S.-L., 2023. MultiEMO: An attention-based correlation-aware multimodal fusion framework for emotion recognition in conversations. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 14752\u201314766.","DOI":"10.18653\/v1\/2023.acl-long.824"},{"key":"10.1016\/j.csl.2026.101965_b45","series-title":"Efficient long-distance latent relation-aware graph neural network for multi-modal emotion recognition in conversations","author":"Shou","year":"2024"},{"key":"10.1016\/j.csl.2026.101965_b46","series-title":"Revisiting multi-modal emotion learning with broad state space models and probability-guidance fusion","author":"Shou","year":"2024"},{"key":"10.1016\/j.csl.2026.101965_b47","doi-asserted-by":"crossref","unstructured":"Song, X., Huang, L., Xue, H., Hu, S., 2022. Supervised Prototypical Contrastive Learning for Emotion Recognition in Conversation. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing. pp. 5197\u20135206.","DOI":"10.18653\/v1\/2022.emnlp-main.347"},{"key":"10.1016\/j.csl.2026.101965_b48","series-title":"Context and system fusion in Post-ASR emotion recognition with large language models","author":"Stepachev","year":"2024"},{"key":"10.1016\/j.csl.2026.101965_b49","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A., 2015. Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 1\u20139.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"10.1016\/j.csl.2026.101965_b50","unstructured":"Tang, C., Yu, W., Sun, G., Chen, X., Tan, T., Li, W., Lu, L., Zejun, M., Zhang, C., 2024. SALMONN: Towards Generic Hearing Abilities for Large Language Models. In: The Twelfth International Conference on Learning Representations."},{"key":"10.1016\/j.csl.2026.101965_b51","series-title":"2024 IEEE Spoken Language Technology Workshop","first-page":"371","article-title":"Large language model based generative error correction: A challenge and baselines for speech recognition, speaker tagging, and emotion recognition","author":"Yang","year":"2024"},{"key":"10.1016\/j.csl.2026.101965_b52","doi-asserted-by":"crossref","unstructured":"Yu, F., Guo, J., Wu, Z., Dai, X., 2024. Emotion-Anchored Contrastive Learning Framework for Emotion Recognition in Conversation. In: Findings of the Association for Computational Linguistics: NAACL 2024. pp. 4521\u20134534.","DOI":"10.18653\/v1\/2024.findings-naacl.282"},{"key":"10.1016\/j.csl.2026.101965_b53","doi-asserted-by":"crossref","unstructured":"Yun, T., Lim, H., Lee, J., Song, M., 2024. TelME: Teacher-leading Multimodal Fusion Network for Emotion Recognition in Conversation. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers). pp. 82\u201395.","DOI":"10.18653\/v1\/2024.naacl-long.5"},{"key":"10.1016\/j.csl.2026.101965_b54","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chen, M., Poria, S., Cambria, E., Morency, L.-P., 2017. Tensor Fusion Network for Multimodal Sentiment Analysis. In: Proceedings of the 2017 Conference on Empirical Methods in Natural Language Processing. pp. 1103\u20131114.","DOI":"10.18653\/v1\/D17-1115"},{"key":"10.1016\/j.csl.2026.101965_b55","series-title":"Mosi: multimodal corpus of sentiment intensity and subjectivity analysis in online opinion videos","author":"Zadeh","year":"2016"},{"key":"10.1016\/j.csl.2026.101965_b56","unstructured":"Zahiri, S.M., Choi, J.D., 2018. Emotion Detection on TV Show Transcripts with Sequence-Based Convolutional Neural Networks.. In: AAAI Workshops. Vol. 18, pp. 44\u201352."},{"key":"10.1016\/j.csl.2026.101965_b57","unstructured":"Zeghidour, N., Teboul, O., de Chaumont Quitry, F., Tagliasacchi, M., 2021. LEAF: A Learnable Frontend for Audio Classification. In: International Conference on Learning Representations."},{"key":"10.1016\/j.csl.2026.101965_b58","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"A MoE multimodal graph attention network framework for multimodal emotion recognition","author":"Zhang","year":"2025"},{"key":"10.1016\/j.csl.2026.101965_b59","series-title":"Improving speech-based emotion recognition with contextual utterance analysis and LLMs","author":"Zhang","year":"2024"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000288?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000288?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:13:43Z","timestamp":1779225223000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000288"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":59,"alternative-id":["S0885230826000288"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101965","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A Mixture-of-Experts model for multimodal emotion recognition in conversations","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101965","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101965"}}