{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:58:36Z","timestamp":1779382716283,"version":"3.53.1"},"reference-count":17,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"7","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,7,1]]},"DOI":"10.1587\/transinf.2024edl8083","type":"journal-article","created":{"date-parts":[[2025,1,9]],"date-time":"2025-01-09T22:13:17Z","timestamp":1736460797000},"page":"841-844","source":"Crossref","is-referenced-by-count":1,"title":["An Interpretable Multi-Level Feature Disentanglement Algorithm for Speech Emotion Recognition"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Huawei","family":"TAO","sequence":"first","affiliation":[{"name":"Key Laboratory of Food Information Processing and Control, Ministry of Education, Henan University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyi","family":"HU","sequence":"additional","affiliation":[{"name":"Key Laboratory of Food Information Processing and Control, Ministry of Education, Henan University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sixian","family":"LI","sequence":"additional","affiliation":[{"name":"Key Laboratory of Food Information Processing and Control, Ministry of Education, Henan University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunhua","family":"ZHU","sequence":"additional","affiliation":[{"name":"Key Laboratory of Food Information Processing and Control, Ministry of Education, Henan University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"LI","sequence":"additional","affiliation":[{"name":"Center for Complexity Science, Henan University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"XIE","sequence":"additional","affiliation":[{"name":"School of Communication Engineering, Nanjing Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] Y. Yang, Y. Zhang, Z. Zhong, W. Dai, Y. Chen, and M. Chen, \u201cIntelligent in-car emotion regulation interaction system based on speech emotion recognition,\u201d 2024 4th International Conference on Computer, Control and Robotics (ICCCR), Shnaghai, China, pp.142-150, 2024. 10.1109\/icccr61138.2024.10585371","DOI":"10.1109\/ICCCR61138.2024.10585371"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] L. Zuo, M.-W. Mak, and Y. Tu, \u201cPromoting independence of depression and speaker features for speaker disentanglement in speech-based depression detection,\u201d ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Seoul, Korea, pp.10191-10195, 2024. 10.1109\/icassp48485.2024.10448231","DOI":"10.1109\/ICASSP48485.2024.10448231"},{"key":"3","doi-asserted-by":"publisher","unstructured":"[3] Z. Aldeneh and E.M. Provost, \u201cYou\u2019re not you when you\u2019re angry: Robust emotion features emerge by recognizing speakers,\u201d IEEE Trans. Affect. Comput., vol.14, no.2, pp.1351-1362, 2023. 10.1109\/taffc.2021.3086050","DOI":"10.1109\/TAFFC.2021.3086050"},{"key":"4","unstructured":"[4] A. Triantafyllopoulos, M. Song, Z. Yang, X. Jing, and B.W. Schuller, \u201cExploring speaker enrolment for few-shot personalisation in emotional vocalisation prediction,\u201d arXiv:2206.06680, 2022. 10.48550\/arXiv.2206.06680"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] I. Gat, H. Aronowitz, W. Zhu, E. Morais, and R. Hoory, \u201cSpeaker normalization for self-supervised speech emotion recognition,\u201d ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Singapore, Singapore, pp.7342-7346, 2022. 10.1109\/icassp43922.2022.9747460","DOI":"10.1109\/ICASSP43922.2022.9747460"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] H. Li, M. Tu, J. Huang, S. Narayanan, and P. Georgiou, \u201cSpeaker-invariant affective representation learning via adversarial training,\u201d ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Barcelona, Spain, pp.7144-7148, 2020. 10.1109\/icassp40776.2020.9054580","DOI":"10.1109\/ICASSP40776.2020.9054580"},{"key":"7","doi-asserted-by":"publisher","unstructured":"[7] C. Lu, Y. Zong, W. Zheng, Y. Li, C. Tang, and B.W. Schuller, \u201cDomain invariant feature learning for speaker-independent speech emotion recognition,\u201d IEEE\/ACM Trans. Audio Speech Lang. Process., vol.30, pp.2217-2230, 2022. 10.1109\/taslp.2022.3178232","DOI":"10.1109\/TASLP.2022.3178232"},{"key":"8","doi-asserted-by":"publisher","unstructured":"[8] X. Chen, X. Xu, J. Chen, Z. Zhang, T. Takiguchi, and E.R. Hancock, \u201cSpeaker-independent emotional voice conversion via disentangled representations,\u201d IEEE Trans. Multimed., vol.25, pp.7480-7493, 2023. 10.1109\/tmm.2022.3222646","DOI":"10.1109\/TMM.2022.3222646"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] J. Gao, D. Chakraborty, H. Tembine, and O. Olaleye, \u201cNonparallel emotional speech conversion,\u201d 20th Annual Conference of the International Speech Communication Association, INTERSPEECH 2019, Graz, Austria, pp.2858-2862, 2019. 10.21437\/interspeech.2019-2878","DOI":"10.21437\/Interspeech.2019-2878"},{"key":"10","unstructured":"[10] Z. Liu, Y. Wang, S. Vaidya, F. Ruehle, J. Halverson, M. Solja\u010di\u0107, T.Y. Hou, and M. Tegmark, \u201cKAN: Kolmogorov-Arnold networks,\u201d arXiv:2404.19756, 2024. 10.48550\/arXiv.2404.19756"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] Z. Ma, Z. Zheng, J. Ye, J. Li, Z. Gao, S. Zhang, and X. Chen, \u201cemotion2vec: Self-supervised pre-training for speech emotion representation,\u201d arxiv:2312.15185, 2023. 10.48550\/arXiv.2312.15185","DOI":"10.18653\/v1\/2024.findings-acl.931"},{"key":"12","doi-asserted-by":"publisher","unstructured":"[12] S. Chen, C. Wang, Z. Chen, Y. Wu, S. Liu, Z. Chen, J. Li, N. Kanda, T. Yoshioka, X. Xiao, J. Wu, L. Zhou, S. Ren, Y. Qian, Y. Qian, J. Wu, M. Zeng, X. Yu, and F. Wei, \u201cWavLM: Large-scale self-supervised pre-training for full stack speech processing,\u201d IEEE J. Sel. Top. Signal Process., vol.16, no.6, pp.1505-1518, 2022. 10.1109\/jstsp.2022.3188113","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"13","doi-asserted-by":"publisher","unstructured":"[13] W. Chen, X. Xing, P. Chen, and X. Xu, \u201cVesper: A compact and effective pretrained model for speech emotion recognition,\u201d IEEE Trans. Affect. Comput., vol.15, no.3, pp.1711-1724, 2024. 10.1109\/taffc.2024.3369726","DOI":"10.1109\/TAFFC.2024.3369726"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] C. Busso, M. Bulut, C.-C. Lee, A. Kazemzadeh, E. Mower, S. Kim, J.N. Chang, S. Lee, and S.S. Narayanan, \u201cIEMOCAP: Interactive emotional dyadic motion capture database,\u201d Language Resources and Evaluation, vol.42, pp.335-359, 2008. 10.1007\/s10579-008-9076-6","DOI":"10.1007\/s10579-008-9076-6"},{"key":"15","doi-asserted-by":"publisher","unstructured":"[15] W. Fan, X. Xu, B. Cai, and X. Xing, \u201cISNet: Individual standardization network for speech emotion recognition,\u201d IEEE\/ACM Trans. Audio Speech Lang. Process., vol.30, pp.1803-1814, 2022. 10.1109\/taslp.2022.3171965","DOI":"10.1109\/TASLP.2022.3171965"},{"key":"16","doi-asserted-by":"publisher","unstructured":"[16] K.L. Ong, C.P. Lee, H.S. Lim, K.M. Lim, and A. Alqahtani, \u201cMaxMViT-MLP: Multiaxis and multiscale vision transformers fusion network for speech emotion recognition,\u201d IEEE Access, vol.12, pp.18237-18250, 2024. 10.1109\/access.2024.3360483","DOI":"10.1109\/ACCESS.2024.3360483"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] B. Yao and W. Shi, \u201cSpeaker-centric multimodal fusion networks for emotion recognition in conversations,\u201d ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Seoul, Korea, pp.8441-8445, 2024. 10.1109\/icassp48485.2024.10447720","DOI":"10.1109\/ICASSP48485.2024.10447720"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/7\/E108.D_2024EDL8083\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T03:35:36Z","timestamp":1751686536000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/7\/E108.D_2024EDL8083\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,1]]},"references-count":17,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edl8083","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,1]]},"article-number":"2024EDL8083"}}