{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T05:09:59Z","timestamp":1764133799917,"version":"build-2065373602"},"reference-count":31,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,10,1]]},"DOI":"10.1587\/transinf.2024edp7161","type":"journal-article","created":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T18:06:11Z","timestamp":1743530771000},"page":"1206-1216","source":"Crossref","is-referenced-by-count":0,"title":["A Multimodal Emotion Recognition Model with Intra-Modal Enhancement and Inter-Modal Interaction"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Zhe","family":"ZHANG","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, North China University of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiding","family":"WANG","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, North China University of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiali","family":"CUI","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, North China University of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"ZHENG","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, North China University of Technology"},{"name":"Key Laboratory of AI and Information Processing, Education Department of Guangxi Zhuang Autonomous Region, Hechi University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] N. Lubis, D. Lestari, S. Sakti, A. Purwarianti, and S. Nakamura, \u201cConstruction of spontaneous emotion corpus from Indonesian TV talk shows and its application on multimodal emotion recognition,\u201d <i>IEICE Trans. Inf. Syst.<\/i>, vol.E101-D, no.8, pp.2092-2100, 2018. 10.1587\/transinf.2017edp7362","DOI":"10.1587\/transinf.2017EDP7362"},{"key":"2","unstructured":"[2] Y. Huang, C. Du, Z. Xue, X. Chen, H. Zhao, and L. Huang, \u201cWhat makes multi-modal learning better than single (provably),\u201d Advances in Neural Information Processing Systems, vol.14, 2021."},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] Y. Mao, G. Liu, X. Wang, W. Gao, and X. Li, \u201cDialogueTRM: Exploring Multi-Modal Emotion Dynamics in Conversations,\u201d Findings of the Association for Computational Linguistics, Findings of ACL: EMNLP 2021, pp.2694-2704, 2021. 10.18653\/v1\/2021.findings-emnlp.229","DOI":"10.18653\/v1\/2021.findings-emnlp.229"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] D. Hazarika, R. Zimmermann, and S. Poria, \u201cMISA: Modality-Invariant and -Specific Representations for Multimodal Sentiment Analysis,\u201d MM 2020 - Proceedings of the 28th ACM International Conference on Multimedia, pp.1122-1131, 2020. 10.1145\/3394171.3413678","DOI":"10.1145\/3394171.3413678"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] J. Hu, Y. Liu, J. Zhao, and Q. Jin, \u201cMMGCN: Multimodal fusion via deep graph convolution network for emotion recognition in conversation,\u201d ACL-IJCNLP 2021 - 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, Proceedings of the Conference, pp.5666-5675, 2021. 10.18653\/v1\/2021.acl-long.440","DOI":"10.18653\/v1\/2021.acl-long.440"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] A. Joshi, A. Bhat, A. Jain, A. Singh, and A. Modi, \u201cCOGMEN: COntextualized GNN based Multimodal Emotion recognitioN,\u201d NAACL 2022 - 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Proceedings of the Conference, pp.4148-4164, 2022. 10.18653\/v1\/2022.naacl-main.306","DOI":"10.18653\/v1\/2022.naacl-main.306"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] A. Zadeh, M. Chen, S. Poria, E. Cambria, and L.-P. Morency, \u201cTensor fusion network for multimodal sentiment analysis,\u201d EMNLP 2017 - Conference on Empirical Methods in Natural Language Processing, Proceedings, pp.1103-1114, 2017. 10.18653\/v1\/d17-1115","DOI":"10.18653\/v1\/D17-1115"},{"key":"8","doi-asserted-by":"publisher","unstructured":"[8] T. Mittal, U. Bhattacharya, R. Chandra, A. Bera, and D. Manocha, \u201cM3ER: Multiplicative multimodal emotion recognition using facial, textual, and speech cues,\u201d AAAI 2020 - 34th AAAI Conference on Artificial Intelligence, vol.34, no.2, pp.1359-1367, 2020. 10.1609\/aaai.v34i02.5492","DOI":"10.1609\/aaai.v34i02.5492"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] K. Chumachenko, A. Iosifidis, and M. Gabbouj, \u201cSelf-attention fusion for audiovisual emotion recognition with incomplete data,\u201d Proceedings - International Conference on Pattern Recognition, pp.2822-2828, 2022. 10.1109\/icpr56361.2022.9956592","DOI":"10.1109\/ICPR56361.2022.9956592"},{"key":"10","doi-asserted-by":"publisher","unstructured":"[10] D. Hu, C. Chen, P. Zhang, J. Li, Y. Yan, and Q. Zhao, \u201cA two-stage attention based modality fusion framework for multi-modal speech emotion recognition,\u201d IEICE Trans. Inf. Syst., vol.E104-D, no.8, pp.1391-1394, 2021. 10.1587\/transinf.2021edl8002","DOI":"10.1587\/transinf.2021EDL8002"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] D. Hu, X. Hou, L. Wei, L. Jiang, and Y. Mo, \u201cMM-DFN: MULTIMODAL DYNAMIC FUSION NETWORK FOR EMOTION RECOGNITION IN CONVERSATIONS,\u201d ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings, pp.7037-7041, 2022. 10.1109\/icassp43922.2022.9747397","DOI":"10.1109\/ICASSP43922.2022.9747397"},{"key":"12","unstructured":"[12] T. Shu, X. Wang, R. Wang, C. Chen, Y. Zhang, and X. Sun, \u201cMutilmodal feature extraction and attention-based fusion for emotion estimation in videos,\u201d arXiv preprint arXiv:2303.10421, 2023."},{"key":"13","doi-asserted-by":"publisher","unstructured":"[13] H. Ma, J. Wang, H. Lin, B. Zhang, Y. Zhang, and B. Xu, \u201cA Transformer-Based Model With Self-Distillation for Multimodal Emotion Recognition in Conversations,\u201d IEEE Trans. Multimedia, vol.26, pp.776-788, 2024. 10.1109\/tmm.2023.3271019","DOI":"10.1109\/TMM.2023.3271019"},{"key":"14","unstructured":"[14] Y. Liu, M. Ott, N. Goyal, J. Du, M. Joshi, D. Chen, O. Levy, M. Lewis, L. Zettlemoyer, and V. Stoyanov, \u201cRoBERTa: A robustly optimized BERT pretraining approach,\u201d arXiv preprint arXiv:1907.11692, 2019."},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] F. Eyben, M. W\u00f6llmer, and B. Schuller, \u201cOpenSMILE: The Munich versatile and fast open-source audio feature extractor,\u201d MM\u201910 - Proceedings of the ACM Multimedia 2010 International Conference, pp.1459-1462, 2010. 10.1145\/1873951.1874246","DOI":"10.1145\/1873951.1874246"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] G. Huang, Z. Liu, L. Van Der Maaten, and K.Q. Weinberger, \u201cDensely connected convolutional networks,\u201d Proceedings - 30th IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, pp.2261-2269, 2017. 10.1109\/cvpr.2017.243","DOI":"10.1109\/CVPR.2017.243"},{"key":"17","doi-asserted-by":"publisher","unstructured":"[17] C. Busso, M. Bulut, C.-C. Lee, A. Kazemzadeh, E. Mower, S. Kim, J.N. Chang, S. Lee, and S.S. Narayanan, \u201cIEMOCAP: Interactive emotional dyadic motion capture database,\u201d Lang Resour Eval, vol.42, no.4, pp.335-359, 2008. 10.1007\/s10579-008-9076-6","DOI":"10.1007\/s10579-008-9076-6"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] S. Poria, D. Hazarika, N. Majumder, G. Naik, E. Cambria, and R. Mihalcea, \u201cMELD: A multimodal multi-party dataset for emotion recognition in conversations,\u201d ACL 2019 - 57th Annual Meeting of the Association for Computational Linguistics, Proceedings of the Conference, pp.527-536, 2020. 10.18653\/v1\/p19-1050","DOI":"10.18653\/v1\/P19-1050"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] R. Azad, L. Niggemeier, M. H\u00fcttemann, A. Kazerouni, E.K. Aghdam, Y. Velichko, U. Bagci, and D. Merhof, \u201cBeyond Self-Attention: Deformable Large Kernel Attention for Medical Image Segmentation,\u201d Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp.1276-1286, 2024. 10.1109\/wacv57701.2024.00132","DOI":"10.1109\/WACV57701.2024.00132"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] K.W. Lau, L.-M. Po, and Y.A.U. Rehman, \u201cLarge Separable Kernel Attention: Rethinking the Large Kernel Attention design in CNN,\u201d Expert Syst. Appl., vol.236, p.121352, 2024. 10.1016\/j.eswa.2023.121352","DOI":"10.1016\/j.eswa.2023.121352"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDeep residual learning for image recognition,\u201d Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, pp.770-778, 2016. 10.1109\/cvpr.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] Y. Zhang, M. Chen, J. Shen, and C. Wang, \u201cTailor Versatile Multi-Modal Learning for Multi-Label Emotion Recognition,\u201d Proceedings of the 36th AAAI Conference on Artificial Intelligence, AAAI 2022, vol.36, no.8, pp.9100-9108, 2022. 10.1609\/aaai.v36i8.20895","DOI":"10.1609\/aaai.v36i8.20895"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] M.A. Manzoor, S. Albarri, Z. Xian, Z. Meng, P. Nakov, and S. Liang, \u201cMultimodality Representation Learning: A Survey on Evolution, Pretraining and Its Applications,\u201d ACM Transactions on Multimedia Computing, Communications and Applications, vol.20, no.3, pp.1-34, 2023. 10.1145\/3617833","DOI":"10.1145\/3617833"},{"key":"24","unstructured":"[24] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, L. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Advances in Neural Information Processing Systems, vol.2017-December, 2017."},{"key":"25","unstructured":"[25] C. Wu, F. Wu, T. Qi, Y. Huang, and X. Xie, \u201cFastformer: Additive attention can be all you need,\u201d arXiv preprint arXiv:2108.09084, 2021."},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] J. Ainslie, J. Lee-Thorp, M. de Jong, Y. Zemlyanskiy, F. Lebron, and S. Sanghai, \u201cGQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints,\u201d EMNLP 2023 - 2023 Conference on Empirical Methods in Natural Language Processing, Proceedings, pp.4895-4901, 2023. 10.18653\/v1\/2023.emnlp-main.298","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"27","unstructured":"[27] A. Paszke, S. Gross, F. Massa, A. Lerer, J. Bradbury Google, G. Chanan, T. Killeen, Z. Lin, N. Gimelshein, L. Antiga, A. Desmaison, A.K. Xamla, E. Yang, Z. Devito, M. Raison Nabla, A. Tejani, S. Chilamkurthy, Q. Ai, B. Steiner, L.F. Facebook, J.B. Facebook, and S. Chintala, \u201cPyTorch: An imperative style, high-performance deep learning library,\u201d Advances in Neural Information Processing Systems, NeurIPS, no.NeurIPS, 2019."},{"key":"28","unstructured":"[28] D.P. Kingma and J.L. Ba, \u201cAdam: A method for stochastic optimization,\u201d 3rd International Conference on Learning Representations, ICLR 2015 - Conference Track Proceedings, 2015."},{"key":"29","doi-asserted-by":"publisher","unstructured":"[29] N. Majumder, S. Poria, D. Hazarika, R. Mihalcea, A. Gelbukh, and E. Cambria, \u201cDialogueRNN: An attentive RNN for emotion detection in conversations,\u201d 33rd AAAI Conference on Artificial Intelligence, AAAI 2019, 31st Innovative Applications of Artificial Intelligence Conference, IAAI 2019 and the 9th AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI 2019, vol.33, no.1, pp.6818-6825, 2019. 10.1609\/aaai.v33i01.33016818","DOI":"10.1609\/aaai.v33i01.33016818"},{"key":"30","doi-asserted-by":"publisher","unstructured":"[30] S.H. Zou, X. Huang, X.D. Shen, and H. Liu, \u201cImproving multimodal fusion with Main Modal Transformer for emotion recognition in conversation,\u201d Knowl. Based Syst., vol.258, p.109978, 2022. 10.1016\/j.knosys.2022.109978","DOI":"10.1016\/j.knosys.2022.109978"},{"key":"31","unstructured":"[31] C.-V.T. Nguyen, C.-B. Nguyen, Q.-T. Ha, and D.-T. Le, \u201cCurriculum learning meets directed acyclic graph for multimodal emotion recognition,\u201d arXiv preprint arXiv:2402.17269, 2024."}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7161\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,4]],"date-time":"2025-10-04T03:28:27Z","timestamp":1759548507000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDP7161\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"references-count":31,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edp7161","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"article-number":"2024EDP7161"}}