{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T14:28:54Z","timestamp":1781188134484,"version":"3.54.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"19","license":[{"start":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T00:00:00Z","timestamp":1677888000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T00:00:00Z","timestamp":1677888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s11042-023-14600-0","type":"journal-article","created":{"date-parts":[[2023,3,4]],"date-time":"2023-03-04T09:02:37Z","timestamp":1677920557000},"page":"28917-28935","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["Multimodal speech emotion recognition based on multi-scale MFCCs and multi-view attention mechanism"],"prefix":"10.1007","volume":"82","author":[{"given":"Lin","family":"Feng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1366-4151","authenticated-orcid":false,"given":"Lu-Yao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sheng-Lan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Han-Qing","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,3,4]]},"reference":[{"issue":"12","key":"14600_CR1","doi-asserted-by":"publisher","first-page":"2423","DOI":"10.1109\/TASLP.2018.2867099","volume":"26","author":"M Abdelwahab","year":"2018","unstructured":"Abdelwahab M, Busso C (2018) Domain adversarial for acoustic emotion recognition. IEEE\/ACM Trans Audio Speech Lang Process 26 (12):2423\u20132435","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"14600_CR2","doi-asserted-by":"crossref","unstructured":"Bhosale S, Chakraborty R, Kopparapu SK (2020) Deep encoder linguistic and acoustic cues for attention based end to end speech emotion recognition. In: ICASSP 2020 - 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7184\u20137188","DOI":"10.1109\/ICASSP40776.2020.9054621"},{"key":"14600_CR3","doi-asserted-by":"crossref","unstructured":"Bird S (2006) NLTK: the natural language toolkit. In: Proceedings of the COLING\/ACL 2006 Interactive Presentation Sessions, pp 69\u201372, Sydney, Australia. Association for computational linguistics","DOI":"10.3115\/1225403.1225421"},{"issue":"4","key":"14600_CR4","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso C, Bulut M, Lee CC, Kazemzadeh A, Mower E, Kim S, Chang JN, Lee S, Narayanan SS (2008) IEMOCAP interactive emotional dyadic motion capture database. Lang Resour Eval 42(4):335\u2013359","journal-title":"Lang Resour Eval"},{"issue":"1","key":"14600_CR5","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1109\/TAFFC.2016.2515617","volume":"8","author":"C Busso","year":"2017","unstructured":"Busso C, Parthasarathy S, Burmania A, Abdelwahab M, Sadoughi N, Provost EM (2017) MSP- IMPROV: an acted corpus of dyadic interactions to study emotion perception. IEEE Trans Affect Comput 8(1):67\u201380","journal-title":"IEEE Trans Affect Comput"},{"key":"14600_CR6","doi-asserted-by":"crossref","unstructured":"Cho J, Pappagari R, Kulkarni P, Villalba J, Carmiel Y, Dehak N (2018) Deep neural networks for emotion recognition combining audio and transcripts. In: Proceedings of the annual conference of the international speech communication association, INTERSPEECH, 2018-September (September), pp 247\u2013251","DOI":"10.21437\/Interspeech.2018-2466"},{"issue":"1-2","key":"14600_CR7","doi-asserted-by":"publisher","first-page":"1261","DOI":"10.1007\/s11042-019-08222-8","volume":"79","author":"F Daneshfar","year":"2020","unstructured":"Daneshfar F, Kabudian SJ (2020) Speech emotion recognition using discriminative dimension reduction by employing a modified quantum-behaved particle swarm optimization algorithm. Multimed Tools Appl 79(1-2):1261\u20131289","journal-title":"Multimed Tools Appl"},{"key":"14600_CR8","doi-asserted-by":"crossref","unstructured":"Dellaert F, Polzin T, Waibel A (1996) Recognizing emotion in speech. In: International conference on spoken language processing, ICSLP, proceedings, vol 3(Icslp 96), pp 1970\u20131973","DOI":"10.21437\/ICSLP.1996-462"},{"issue":"1","key":"14600_CR9","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1109\/TASLP.2017.2759338","volume":"26","author":"J Deng","year":"2018","unstructured":"Deng J, Xu X, Zhang Z, Fruhholz S, Schuller B (2018) Semisupervised autoencoders for speech emotion recognition. IEEE\/ACM Trans Audio Speech Lang Process 26(1):31\u201343","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"14600_CR10","doi-asserted-by":"crossref","unstructured":"Dobri\u0161ek S, Gaj\u0161ek R, Miheli\u010d F, Pave\u0161i\u0107 N, \u0160truc V (2013) Towards efficient multi-modal emotion recognition. Int J Adv Robot Syst, vol 10","DOI":"10.5772\/54002"},{"key":"14600_CR11","doi-asserted-by":"crossref","unstructured":"Georgiou E, Papaioannou C, Potamianos A (2019) Deep hierarchical fusion with application in sentiment analysis. In: Proceedings of the annual conference of the international speech communication association, INTERSPEECH, 2019-September, pp 1646\u20131650","DOI":"10.21437\/Interspeech.2019-3243"},{"issue":"1-2","key":"14600_CR12","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/S0167-6393(02)00082-1","volume":"40","author":"C Gobl","year":"2003","unstructured":"Gobl C, Chasaide AN (2003) The role of voice quality in communicating emotion, mood and attitude. Speech Comm 40(1-2):189\u2013212","journal-title":"Speech Comm"},{"key":"14600_CR13","doi-asserted-by":"crossref","unstructured":"Guizzo E, Weyde T, Leveson JB (2020) Multi-time-scale convolution for emotion recognition from speech audio signals. In: ICASSP 2020 \u2013 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6489\u20136493","DOI":"10.1109\/ICASSP40776.2020.9053727"},{"key":"14600_CR14","doi-asserted-by":"crossref","unstructured":"Gupta S, fahad MS, Deepak A (2019) Pitch-synchronous single frequency filtering spectrogram for speech emotion recognition. arXiv, pp 23347\u201323365","DOI":"10.1007\/s11042-020-09068-1"},{"issue":"JUN","key":"14600_CR15","first-page":"1","volume":"9","author":"A Lausen","year":"2018","unstructured":"Lausen A, Schacht A (2018) Gender differences in the recognition of vocal emotions. Front Psychol 9(JUN):1\u201322","journal-title":"Front Psychol"},{"key":"14600_CR16","doi-asserted-by":"crossref","unstructured":"Li P, Song Y, McLoughlin IV, Guo W, Dai L (2018) An attention pooling based representation learning method for speech emotion recognition. In: Interspeech 2018, 19th annual conference of the international speech communication association. ISCA, Hyderabad, India, 2-6 Sept 2018, pp 3087\u20133091","DOI":"10.21437\/Interspeech.2018-1242"},{"key":"14600_CR17","doi-asserted-by":"crossref","unstructured":"Liu J, Liu Z, Wang L, Guo L, Dang J (2020) Speech emotion recognition with local-global aware deep representation learning. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp 7169\u20137173","DOI":"10.1109\/ICASSP40776.2020.9053192"},{"key":"14600_CR18","doi-asserted-by":"crossref","unstructured":"Liu Z, Shen Y, Lakshminarasimhan VB, Liang PP, Zadeh A, Morency LP (2018) Efficient low-rank multimodal fusion with modality-specific factors. arXiv, pp 2247\u20132256","DOI":"10.18653\/v1\/P18-1209"},{"key":"14600_CR19","doi-asserted-by":"crossref","unstructured":"Metallinou A, Wollmer M, Katsamanis A, Eyben F, Schuller B, Narayanan S (2012) Context-sensitive learning for enhanced audiovisual emotion classification. IEEE Trans Affect Comput:184\u2013198","DOI":"10.1109\/T-AFFC.2011.40"},{"key":"14600_CR20","doi-asserted-by":"crossref","unstructured":"Miao H, Cheng G, Gao C, Zhang P, online YY (2020) Transformer-based ctc\/attention end-to-end speech recognition architecture. In: ICASSP 2020 - 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6084\u20136088","DOI":"10.1109\/ICASSP40776.2020.9053165"},{"key":"14600_CR21","unstructured":"Mihalcea R, Morency L-P, Science C (2013) Utterance-level multimodal sentiment analysis. Acl: 973\u2013982"},{"key":"14600_CR22","doi-asserted-by":"crossref","unstructured":"Nediyanchath A, Paramasivam P, Yenigalla P (2020) Multi-head attention for speech emotion recognition with auxiliary learning of gender recognition. In: ICASSP, IEEE international conference on acoustics, speech and signal processing - proceedings. IEEE, vol 2020 May, pp 7179\u20137183","DOI":"10.1109\/ICASSP40776.2020.9054073"},{"key":"14600_CR23","first-page":"809","volume":"2","author":"D Neiberg","year":"2006","unstructured":"Neiberg D, Elenius K, Laskowski K (2006) Emotion recognition in spontaneous speech using GMMs. Proc Annual Conf Int Speech Commun Assoc Interspeech 2:809\u2013812","journal-title":"Proc Annual Conf Int Speech Commun Assoc Interspeech"},{"key":"14600_CR24","doi-asserted-by":"crossref","unstructured":"Neumann M, Vu NT (2019) Improving speech emotion recognition with unsupervised representation learning on unlabeled speech. In: ICASSP 2019 - 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7390\u20137394","DOI":"10.1109\/ICASSP.2019.8682541"},{"key":"14600_CR25","doi-asserted-by":"crossref","unstructured":"Pao T-L, Chen Y-T, Yeh J-H, Liao W-Y (2005) Combining acoustic features for improved emotion recognition in mandarin speech. In: Tao J, Tan T, Picard RW (eds) Affective computing and intelligent interaction, pp 279\u2013285, Berlin, Heidelberg, Springer","DOI":"10.1007\/11573548_36"},{"key":"14600_CR26","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.inffus.2017.02.003","volume":"37","author":"S Poria","year":"2017","unstructured":"Poria S, Cambria E, Bajpai R, Hussain A (2017) A review of affective computing: from unimodal analysis to multimodal fusion . Inf Fusion 37:98\u2013125","journal-title":"Inf Fusion"},{"key":"14600_CR27","doi-asserted-by":"crossref","unstructured":"Poria S, Cambria E, Gelbukh A (2015) Deep convolutional neural network textual features and multiple kernel learning for utterance-level multimodal sentiment analysis. In: Conference proceedings - EMNLP 2015: conference on empirical methods in natural language processing, (september), pp 2539\u20132544","DOI":"10.18653\/v1\/D15-1303"},{"key":"14600_CR28","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1016\/j.neucom.2015.01.095","volume":"174","author":"S Poria","year":"2016","unstructured":"Poria S, Cambria E, Howard N, Huang GB, Hussain A (2016) Fusing audio, visual and textual clues for sentiment analysis from multimodal content. Neurocomputing 174:50\u201359","journal-title":"Neurocomputing"},{"key":"14600_CR29","unstructured":"Rozgi\u0107 V, Ananthakrishnan S, Saleem S, Kumar R, Prasad R (2012) Ensemble of SVM trees for multimodal emotion recognition. In: 2012 Conference handbook - asia-pacific signal and information processing association annual summit and conference, APSIPA ASC 2012, pp 7\u201310"},{"issue":"4","key":"14600_CR30","doi-asserted-by":"publisher","first-page":"192","DOI":"10.1109\/T-AFFC.2011.17","volume":"2","author":"B Schuller","year":"2011","unstructured":"Schuller B (2011) Recognizing affect from linguistic information in 3D continuous space. IEEE Trans Affect Comput 2(4):192\u2013205","journal-title":"IEEE Trans Affect Comput"},{"key":"14600_CR31","doi-asserted-by":"crossref","unstructured":"Shen J, Tang X, Dong X, Shao L (2020) Visual object tracking by hierarchical attention siamese network, vol 50(7), pp 3068\u20133080","DOI":"10.1109\/TCYB.2019.2936503"},{"key":"14600_CR32","doi-asserted-by":"crossref","unstructured":"Shirian A, Guha T (2020) Compact graph architecture for speech emotion recognition","DOI":"10.1109\/ICASSP39728.2021.9413876"},{"key":"14600_CR33","doi-asserted-by":"crossref","unstructured":"Su B-H, Chang C-M, Lin Y-S, Lee C-C (2020) Improving speech emotion recognition using graph attentive bi-directional gated recurrent unit network, pp 506\u2013510","DOI":"10.21437\/Interspeech.2020-1733"},{"key":"14600_CR34","unstructured":"Tripathi S, Tripathi S, Beigi H (2018) Multi-modal emotion recognition on IEMOCAP dataset using deep learning"},{"key":"14600_CR35","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Guyon I, Luxburg UV, Bengio S, Wallach H, Fergus R, Vishwanathan S, Garnett R (eds) Advances in neural information processing systems. Curran Associates, Inc, vol 30, pp 5998\u20136008"},{"key":"14600_CR36","doi-asserted-by":"crossref","unstructured":"Wang Y, Guan L (2004) An investigation of speech-based human emotion recognition. In: 2004 IEEE 6th workshop on multimedia signal processing, pp 15\u201318","DOI":"10.1109\/MMSP.2004.1436403"},{"key":"14600_CR37","doi-asserted-by":"crossref","unstructured":"Wang Y, Shen Y, Liu Z, Liang PP, Zadeh A, Morency LP (2018) Words can Shift: dynamically adjusting word representations using nonverbal behaviors. arXiv","DOI":"10.1609\/aaai.v33i01.33017216"},{"key":"14600_CR38","doi-asserted-by":"crossref","unstructured":"Wang J, Xue M, Culhane R, Diao E, Ding J, Tarokh V (2020) Speech emotion recognition with dual-sequence LSTM architecture. In: ICASSP 2020\u20132020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6469\u20136473","DOI":"10.1109\/ICASSP40776.2020.9054629"},{"key":"14600_CR39","unstructured":"Williams C, Stevens K (1981) Vocal correlates of emotional state. Vocal Correlates Emotional States, vol 01"},{"key":"14600_CR40","doi-asserted-by":"crossref","unstructured":"Wu CH, Liang WB (2015) Emotion recognition of affective speech based on multiple classifiers using acoustic-prosodic information and semantic labels (Extended abstract). In: 2015 International conference on affective computing and intelligent interaction, ACII 2015, pp 477\u2013483","DOI":"10.1109\/ACII.2015.7344613"},{"key":"14600_CR41","doi-asserted-by":"crossref","unstructured":"Xu H, Zhang H, Han K, Wang Y, Peng Y, Li X (2019) Learning alignment for multimodal emotion recognition from speech. arXiv, pp 3569\u20133573","DOI":"10.21437\/Interspeech.2019-3247"},{"key":"14600_CR42","doi-asserted-by":"crossref","unstructured":"Yandong W, Kaipeng Z, Zhifeng L, Yu Q (2016) A Discriminative Feature Learning Approach for Deep Face Recognition. Computer Vision \u2013 ECCV pp 499\u2013515","DOI":"10.1007\/978-3-319-46478-7_31"},{"key":"14600_CR43","doi-asserted-by":"crossref","unstructured":"Yoon S, Byun S, Jung K (2018) Multimodal speech emotion recognition using audio and text. 2018 IEEE Spoken Lang Technol Workshop (SLT):112\u2013118","DOI":"10.1109\/SLT.2018.8639583"},{"key":"14600_CR44","doi-asserted-by":"crossref","unstructured":"Yoon S, Dey S, Lee H, Jung K (2020) Attention modality hopping mechanism for speech emotion recognition. In: ICASSP 2020 - 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 3362\u20133366","DOI":"10.1109\/ICASSP40776.2020.9054229"},{"key":"14600_CR45","doi-asserted-by":"crossref","unstructured":"Zhang Z, Wu B, Schuller B (2019) Attention-augmented end-to-end multi-task learning for emotion prediction from speech. In: ICASSP 2019 - 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6705\u20136709","DOI":"10.1109\/ICASSP.2019.8682896"},{"key":"14600_CR46","doi-asserted-by":"crossref","unstructured":"Zhao Z, Bao Z, Zhang Z, Cummins N, Wang H, Schuller BW (2019) Attention-enhanced connectionist temporal classification for discrete speech emotion recognition. In: Interspeech 2019, 20th annual conference of the international speech communication association. ISCA, Graz, Austria, 15-19 Sept 2019, pp 206\u2013210","DOI":"10.21437\/Interspeech.2019-1649"},{"key":"14600_CR47","doi-asserted-by":"crossref","unstructured":"Zheng WQ, Yu JS, Zou YX (2015) An experimental study of speech emotion recognition based on deep convolutional neural networks. In: 2015 International conference on affective computing and intelligent interaction, ACII 2015, pp 827\u2013831","DOI":"10.1109\/ACII.2015.7344669"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14600-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-14600-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14600-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,22]],"date-time":"2023-07-22T10:34:14Z","timestamp":1690022054000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-14600-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,4]]},"references-count":47,"journal-issue":{"issue":"19","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["14600"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-14600-0","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3,4]]},"assertion":[{"value":"7 September 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 February 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}