{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T15:43:08Z","timestamp":1764603788421},"reference-count":14,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"8","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2021,8,1]]},"DOI":"10.1587\/transinf.2021edl8002","type":"journal-article","created":{"date-parts":[[2021,7,31]],"date-time":"2021-07-31T22:16:48Z","timestamp":1627769808000},"page":"1391-1394","source":"Crossref","is-referenced-by-count":4,"title":["A Two-Stage Attention Based Modality Fusion Framework for Multi-Modal Speech Emotion Recognition"],"prefix":"10.1587","volume":"E104.D","author":[{"given":"Dongni","family":"HU","sequence":"first","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"},{"name":"University of Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengxin","family":"CHEN","sequence":"additional","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"},{"name":"University of Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengyuan","family":"ZHANG","sequence":"additional","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"},{"name":"University of Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junfeng","family":"LI","sequence":"additional","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"},{"name":"University of Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghong","family":"YAN","sequence":"additional","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"},{"name":"University of Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingwei","family":"ZHAO","sequence":"additional","affiliation":[{"name":"Key Laboratory of Speech Acoustics and Content Understanding, Institute of Acoustics, Chinese Academy of Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] B. Schuller, \u201cSpeech emotion recognition: Two decades in a nutshell benchmarks and ongoing trends,\u201d Commun. ACM, vol.61, no.5, pp.90-99, April 2018. 10.1145\/3129340","DOI":"10.1145\/3129340"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] S. Poria, E. Cambria, R. Bajpai, and A. Hussain, \u201cA Review of Affective Computing: From Unimodal Analysis to Multimodal Fusion,\u201d Information Fusion, vol.37, pp.98-125, Sept. 2017. 10.1016\/j.inffus.2017.02.003","DOI":"10.1016\/j.inffus.2017.02.003"},{"key":"3","unstructured":"[3] S. Tripathi and H. Beigi, \u201cMulti-modal emotion recognition on iemocap dataset using deep learning,\u201d arXiv preprint arXiv:1804.05788, 2019."},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] E. Georgiou, C. Papaioannou, and A. Potamianos, \u201cDeep hierarchical fusion with application in sentiment analysis,\u201d Proc. Interspeech 2019, pp.1646-1650, 2019. 10.21437\/Interspeech.2019-3243","DOI":"10.21437\/Interspeech.2019-3243"},{"key":"5","doi-asserted-by":"publisher","unstructured":"[5] S. Poria, N. Majumder, D. Hazarika, E. Cambria, A. Gelbukh, and A. Hussain, \u201cMultimodal sentiment analysis: Addressing key issues and setting up the baselines,\u201d IEEE Intelligent Systems, vol.33, no.6, pp.17-25, Nov.-Dec. 2018. 10.1109\/MIS.2018.2882362","DOI":"10.1109\/MIS.2018.2882362"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] S. Yoon, S. Byun, and K. Jung, \u201cMultimodal speech emotion recognition using audio and text,\u201d 2018 IEEE Spoken Language Technology Workshop (SLT), 2018. 10.1109\/SLT.2018.8639583","DOI":"10.1109\/SLT.2018.8639583"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] M. Chen and X. Zhao, \u201cA multi-scale fusion framework for bimodal speech emotion recognition,\u201d Proc. Interspeech, pp.374-378, 2020. 10.21437\/Interspeech.2020-3156","DOI":"10.21437\/Interspeech.2020-3156"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] H. Xu, H. Zhang, K. Han, Y. Wang, Y. Peng, and X. Li, \u201cLearning alignment for multimodal emotion recognition from speech[J],\u201d arXiv preprint arXiv:1909.05645, 2019. 10.21437\/Interspeech.2019-3247","DOI":"10.21437\/Interspeech.2019-3247"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] S. Yoon, S. Byun, S. Dey, and K. Jung, \u201cSpeech emotion recognition using multi-hop attention mechanism,\u201d ICASSP 2019-2019 IEEE Int. Conf. Acoust., Speech, Signal Process. (ICASSP), pp.2822-2826, 2019. 10.1109\/ICASSP.2019.8683483","DOI":"10.1109\/ICASSP.2019.8683483"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] Z. Pan, Z. Luo., J. Yang, and H. Li, \u201cMulti-modal Attention for Speech Emotion Recognition,\u201d arXiv preprint arXiv:2009.04107 (2020). 10.21437\/Interspeech.2020-1653","DOI":"10.21437\/Interspeech.2020-1653"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] P. Gao, Z. Jiang, H. You, P. Lu, S.C.H. Hoi, X. Wang, and H. Li, \u201cDynamic fusion with intra- and inter-modality attention flow for visual question answering,\u201d Proc. IEEE Conf. Comput. Vis. Pattern Recognit., 2019. 10.1109\/CVPR.2019.00680","DOI":"10.1109\/CVPR.2019.00680"},{"key":"12","doi-asserted-by":"publisher","unstructured":"[12] C. Busso, M. Bulut, C.-C. Lee, A. Kazemzadeh, E. Mower, S. Kim, J.N. Chang, S. Lee, and S.S. Narayanan, \u201cIEMOCAP: Interactive emotional dyadic motion capture database,\u201d Language resources and evaluation, vol.42, no.4, p.335, 2008. 10.1007\/s10579-008-9076-6","DOI":"10.1007\/s10579-008-9076-6"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] J. Pennington, R. Socher, and C. Manning, \u201cGlove: Global vectors for word representation,\u201d Proc. Conf. Empirical Methods in Natural Language Processing, pp.1532-1543, Oct. 2014. 10.3115\/v1\/D14-1162","DOI":"10.3115\/v1\/D14-1162"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] B.W. Schuller, S. Steidl, and A. Batliner, \u201cThe INTERSPEECH 2009 emotion challenge,\u201d Interspeech 2009, 10th Annual Conference of the International Speech Communication Association, Brighton, United Kingdom, Sept. 6-10, 2009, pp.312-315, 2009.","DOI":"10.21437\/Interspeech.2009-103"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E104.D\/8\/E104.D_2021EDL8002\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,8]],"date-time":"2024-05-08T05:07:07Z","timestamp":1715144827000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E104.D\/8\/E104.D_2021EDL8002\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,8,1]]},"references-count":14,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2021]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2021edl8002","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,8,1]]},"article-number":"2021EDL8002"}}