{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T13:55:37Z","timestamp":1781358937621,"version":"3.54.1"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"17-18","license":[{"start":{"date-parts":[[2024,6,27]],"date-time":"2024-06-27T00:00:00Z","timestamp":1719446400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,27]],"date-time":"2024-06-27T00:00:00Z","timestamp":1719446400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Key Cooperation Project of the Chongqing Municipal Education Commission","award":["HZ2021008"],"award-info":[{"award-number":["HZ2021008"]}]},{"name":"Research Project of Graduate Education and Teaching Reform of Chongqing Municipal Education Commissio","award":["yjg223087"],"award-info":[{"award-number":["yjg223087"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10489-024-05630-8","type":"journal-article","created":{"date-parts":[[2024,6,27]],"date-time":"2024-06-27T03:40:43Z","timestamp":1719459643000},"page":"8478-8490","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Multi-level attention fusion network assisted by relative entropy alignment for multimodal speech emotion recognition"],"prefix":"10.1007","volume":"54","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1965-3170","authenticated-orcid":false,"given":"Jianjun","family":"Lei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,6,27]]},"reference":[{"issue":"7","key":"5630_CR1","doi-asserted-by":"publisher","first-page":"7647","DOI":"10.1007\/s10489-022-03907-4","volume":"53","author":"A Pradhan","year":"2023","unstructured":"Pradhan A, Senapati MR, Sahu PK (2023) A multichannel embedding and arithmetic optimized stacked bi-gru model with semantic attention to detect emotion over text data. Appl Intell 53(7):7647\u20137664","journal-title":"Appl Intell"},{"issue":"8","key":"5630_CR2","doi-asserted-by":"publisher","first-page":"5543","DOI":"10.1007\/s10489-020-02125-0","volume":"51","author":"S Saurav","year":"2021","unstructured":"Saurav S, Saini R, Singh S (2021) Emnet: a deep integrated convolutional neural network for facial emotion recognition in the wild. Appl Intell 51(8):5543\u20135570","journal-title":"Appl Intell"},{"issue":"11","key":"5630_CR3","doi-asserted-by":"publisher","first-page":"14470","DOI":"10.1007\/s10489-022-04216-6","volume":"53","author":"Z Fang","year":"2023","unstructured":"Fang Z, Liu Z, Hung CC, Sekhavat YA, Liu T, Wang X (2023) Learning coordinated emotion representation between voice and face. Appl Intell 53(11):14470\u201314492","journal-title":"Appl Intell"},{"key":"5630_CR4","doi-asserted-by":"crossref","unstructured":"Lieskovsk\u00e1 E, Jakubec M, Jarina R, Chmulik M(2021) A review on speech emotion recognition using deep learning and attention mechanism. Electronics 10(10)","DOI":"10.3390\/electronics10101163"},{"key":"5630_CR5","doi-asserted-by":"publisher","first-page":"44","DOI":"10.1016\/j.specom.2020.11.005","volume":"126","author":"L Alhinti","year":"2021","unstructured":"Alhinti L, Christensen H, Cunningham S (2021) Acoustic differences in emotional speech of people with dysarthria. Speech Comm 126:44\u201360","journal-title":"Speech Comm"},{"key":"5630_CR6","doi-asserted-by":"crossref","unstructured":"Cao Q, Hou M, Chen B, Zhang Z, Lu G (2021) Hierarchical network based on the fusion of static and dynamic features for speech emotion recognition. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6334\u20136338","DOI":"10.1109\/ICASSP39728.2021.9414540"},{"key":"5630_CR7","doi-asserted-by":"publisher","first-page":"51231","DOI":"10.1109\/ACCESS.2021.3069818","volume":"9","author":"C Zhang","year":"2021","unstructured":"Zhang C, Xue L (2021) Autoencoder with emotion embedding for speech emotion recognition. IEEE Access 9:51231\u201351241","journal-title":"IEEE Access"},{"key":"5630_CR8","doi-asserted-by":"crossref","unstructured":"Yin Y, Gu Y, Yao L, Zhou Y, Liang X, Zhang H (2021) Progressive co-teaching for ambiguous speech emotion recognition. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6264\u20136268","DOI":"10.1109\/ICASSP39728.2021.9414494"},{"key":"5630_CR9","doi-asserted-by":"crossref","unstructured":"Shirian A, Guha T (2021) Compact graph architecture for speech emotion recognition. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6284\u20136288","DOI":"10.1109\/ICASSP39728.2021.9413876"},{"key":"5630_CR10","doi-asserted-by":"crossref","unstructured":"Tzirakis P, Nguyen A, Zafeiriou S, Schuller BW (2021) Speech emotion recognition using semantic information. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6279\u20136283","DOI":"10.1109\/ICASSP39728.2021.9414866"},{"key":"5630_CR11","doi-asserted-by":"crossref","unstructured":"Zou H, Si Y, Chen C, Rajan D, Chng ES (2022) Speech emotion recognition with co-attention based multi-level acoustic information. In: ICASSP 2022 - 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7367\u20137371","DOI":"10.1109\/ICASSP43922.2022.9747095"},{"key":"5630_CR12","first-page":"364","volume":"2020","author":"Z Pan","year":"2020","unstructured":"Pan Z, Luo Z, Yang J, Li H (2020) Multi-modal attention for speech emotion recognition. Proc. Interspeech 2020:364\u2013368","journal-title":"Proc. Interspeech"},{"key":"5630_CR13","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"5630_CR14","first-page":"379","volume":"2020","author":"P Liu","year":"2020","unstructured":"Liu P, Li K, Meng H (2020) Group gated fusion on attention-based bidirectional alignment for multimodal emotion recognition. Proc. Interspeech 2020:379\u2013383","journal-title":"Proc. Interspeech"},{"key":"5630_CR15","doi-asserted-by":"publisher","first-page":"3569","DOI":"10.21437\/Interspeech.2019-3247","volume":"2019","author":"H Xu","year":"2019","unstructured":"Xu H, Zhang H, Han K, Wang Y, Peng Y, Li X (2019) Learning alignment for multimodal emotion recognition from speech. Proc. Interspeech 2019:3569\u20133573","journal-title":"Proc. Interspeech"},{"issue":"2","key":"5630_CR16","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/MMUL.2022.3161411","volume":"29","author":"L Guo","year":"2022","unstructured":"Guo L, Wang L, Dang J, Fu Y, Liu J, Ding S (2022) Emotion recognition with multimodal transformer fusion framework based on acoustic and lexical information. IEEE MultiMed 29(2):94\u2013103","journal-title":"IEEE MultiMed"},{"key":"5630_CR17","doi-asserted-by":"crossref","unstructured":"Chen W, Xing X, Xu X, Yang J, Pang J (2022) Key-sparse transformer for multimodal speech emotion recognition. In: ICASSP 2022 - 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6897\u20136901","DOI":"10.1109\/ICASSP43922.2022.9746598"},{"key":"5630_CR18","doi-asserted-by":"crossref","unstructured":"Jiang W, Wang Z, Jin JS, Han X, Li C (2019) Speech emotion recognition with heterogeneous feature unification of deep neural network. Sensors 19(12)","DOI":"10.3390\/s19122730"},{"key":"5630_CR19","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/j.specom.2021.05.009","volume":"132","author":"AR Avila","year":"2021","unstructured":"Avila AR, O\u2019Shaughnessy D, Falk TH (2021) Automatic speaker verification from affective speech using gaussian mixture model based estimation of neutral speech characteristics. Speech Comm 132:21\u201331","journal-title":"Speech Comm"},{"key":"5630_CR20","doi-asserted-by":"crossref","unstructured":"Younis EMG, Zaki SM, Kanjo E, Houssein EH (2022) Evaluating ensemble learning methods for multi-modal emotion recognition using sensor data fusion. Sensors 22(15)","DOI":"10.3390\/s22155611"},{"key":"5630_CR21","doi-asserted-by":"crossref","unstructured":"Xu M, Zhang F, Khan SU (2020) Improve accuracy of speech emotion recognition with attention head fusion. In: 2020 10th Annual computing and communication workshop and conference (CCWC), pp 1058\u20131064","DOI":"10.1109\/CCWC47524.2020.9031207"},{"key":"5630_CR22","doi-asserted-by":"crossref","unstructured":"Xu M, Zhang F, Xiaodong Cui, and Wei Zhang (2021) Speech emotion recognition with multiscale area attention and data augmentation. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6319\u20136323","DOI":"10.1109\/ICASSP39728.2021.9414635"},{"key":"5630_CR23","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1016\/j.neunet.2022.09.022","volume":"156","author":"J Lei","year":"2022","unstructured":"Lei J, Zhu X, Wang Y (2022) Bat: Block and token self-attention for speech emotion recognition. Neural Netw 156:67\u201380","journal-title":"Neural Netw"},{"key":"5630_CR24","first-page":"1748","volume":"2021","author":"P Kumar","year":"2021","unstructured":"Kumar P, Kaushik V, Raman B (2021) Towards the explainability of multimodal speech emotion recognition. Proc Interspeech 2021:1748\u20131752","journal-title":"Proc Interspeech"},{"key":"5630_CR25","doi-asserted-by":"crossref","unstructured":"Chao Li, Zhongtian Bao, Linhao Li, and Ziping Zhao (2020) Exploring temporal representations by leveraging attention-based bidirectional lstm-rnns for multi-modal emotion recognition. Inf Process Manag 57(3)","DOI":"10.1016\/j.ipm.2019.102185"},{"key":"5630_CR26","first-page":"3375","volume":"2021","author":"H Li","year":"2021","unstructured":"Li H, Ding W, Wu Z, Liu Z (2021) Learning Fine-Grained Cross Modality Excitement for Speech Emotion Recognition. Proc Interspeech 2021:3375\u20133379","journal-title":"Proc Interspeech"},{"key":"5630_CR27","doi-asserted-by":"crossref","unstructured":"Peng Z, Lu Y, Pan S, Liu Y (2021) Efficient speech emotion recognition using multi-scale cnn and attention. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 3020\u20133024","DOI":"10.1109\/ICASSP39728.2021.9414286"},{"key":"5630_CR28","unstructured":"Baevski A, Schneider S, Auli M (2020) vq-wav2vec: Self-supervised learning of discrete speech representations. In: International conference on learning representations"},{"key":"5630_CR29","first-page":"3755","volume":"2020","author":"S Siriwardhana","year":"2020","unstructured":"Siriwardhana S, Reis A, Weerasekera R, Nanayakkara S (2020) Jointly fine-tuning \u201cBERT-Like\u2019\u2019 self supervised models to improve multimodal speech emotion recognition. Proc Interspeech 2020:3755\u2013375","journal-title":"Proc Interspeech"},{"issue":"6","key":"5630_CR30","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"Chen S, Wang C, Chen Z, Wu Y, Liu S, Chen Z, Li J, Kanda N, Yoshioka T, Xiao X, Wu J, Zhou L, Ren S, Qian Y, Qian Y, Zeng M, Yu X, Wei F (2022) Wavlm: Large-scale self-supervised pre-training for full stack speech processing. IEEE J Select Top Signal Process 16(6):1505\u20131518","journal-title":"IEEE J Select Top Signal Process"},{"key":"5630_CR31","first-page":"5776","volume":"33","author":"W Wang","year":"2020","unstructured":"Wang W, Wei F, Dong L, Bao H, Yang N, Zhou M (2020) Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Adv Neural Inf Process Syst 33:5776\u20135788","journal-title":"Adv Neural Inf Process Syst"},{"key":"5630_CR32","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD (2014) GloVe: Global vectors for word representation. In: Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP), pp 1532\u20131543, Doha, Qatar, October. Association for Computational Linguistics","DOI":"10.3115\/v1\/D14-1162"},{"issue":"4","key":"5630_CR33","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"Carlos Busso","year":"2008","unstructured":"Busso Carlos, Bulut Murtaza, Lee Chi-Chun, Kazemzadeh Abe, Mower Emily, Kim Samuel, Chang Jeannette N, Lee Sungbok, Narayanan Shrikanth S (2008) Iemocap: interactive emotional dyadic motion capture database. Lang Resour Eval 42(4):335\u2013359","journal-title":"Lang Resour Eval"},{"key":"5630_CR34","doi-asserted-by":"crossref","unstructured":"Poria S, Hazarika D, Majumder N, Naik G, Cambria E, Mihalcea R (2019) MELD: A multimodal multi-party dataset for emotion recognition in conversations. In: Proceedings of the 57th annual meeting of the association for computational linguistics, pp 527\u2013536, Florence, Italy, Association for Computational Linguistics","DOI":"10.18653\/v1\/P19-1050"},{"key":"5630_CR35","first-page":"4243","volume":"2020","author":"DN Krishna","year":"2020","unstructured":"Krishna DN, Patil A (2020) Multimodal emotion recognition using cross-modal attention and 1d convolutional neural networks. Proc Interspeech 2020:4243\u20134247","journal-title":"Proc Interspeech"},{"key":"5630_CR36","doi-asserted-by":"crossref","unstructured":"Makiuchi MR, Uto K, Shinoda K (2021) Multimodal emotion recognition with high-level speech and text features. In: 2021 IEEE automatic speech recognition and understanding workshop (ASRU), pp 350\u2013357","DOI":"10.1109\/ASRU51503.2021.9688036"},{"key":"5630_CR37","doi-asserted-by":"crossref","unstructured":"Nediyanchath A, Paramasivam P, Yenigalla P (2020) Multi-head attention for speech emotion recognition with auxiliary learning of gender recognition. In: ICASSP 2020 - 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7179\u20137183","DOI":"10.1109\/ICASSP40776.2020.9054073"},{"key":"5630_CR38","doi-asserted-by":"crossref","unstructured":"Liu Y, Sun H, Guan W, Xia Y, Zhao Z (2022) Multi-modal speech emotion recognition using self-attention mechanism and multi-scale fusion framework. Speech Comm 139:1\u20139","DOI":"10.1016\/j.specom.2022.02.006"},{"key":"5630_CR39","doi-asserted-by":"crossref","unstructured":"Feng L, Liu LY, Liu SL, Zhou J, Yang HQ, Yang J (2023) Multimodal speech emotion recognition based on multi-scale mfccs and multi-view attention mechanism. Multimed Tools Appl 2023","DOI":"10.1007\/s11042-023-14600-0"},{"key":"5630_CR40","doi-asserted-by":"crossref","unstructured":"Liang J, Li R, Jin Q (2020) Semi-supervised multi-modal emotion recognition with cross-modal distribution matching. In: Proceedings of the 28th ACM international conference on multimedia, MM \u201920, page 2852\u20132861, New York, NY, USA. Association for Computing Machinery","DOI":"10.1145\/3394171.3413579"},{"key":"5630_CR41","doi-asserted-by":"publisher","first-page":"629","DOI":"10.1016\/j.neucom.2022.06.072","volume":"501","author":"Y Shou","year":"2022","unstructured":"Shou Y, Meng T, Ai W, Yang S, Li K (2022) Conversational emotion recognition studies based on graph convolutional neural networks and a dependent syntactic analysis. Neurocomputing 501:629\u2013639","journal-title":"Neurocomputing"},{"key":"5630_CR42","doi-asserted-by":"crossref","unstructured":"Sun L, Liu B, Tao J, Lian Z (2021) Multimodal cross- and self-attention network for speech emotion recognition. In: ICASSP 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4275\u20134279","DOI":"10.1109\/ICASSP39728.2021.9414654"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05630-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05630-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05630-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,7]],"date-time":"2024-08-07T12:37:59Z","timestamp":1723034279000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05630-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,27]]},"references-count":42,"journal-issue":{"issue":"17-18","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["5630"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05630-8","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6,27]]},"assertion":[{"value":"18 June 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 June 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}},{"value":"The manuscript was reviewed and ethical approved for publication by all authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical conduct"}}]}}