{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T14:02:14Z","timestamp":1762351334731,"version":"3.37.3"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,8,28]],"date-time":"2022-08-28T00:00:00Z","timestamp":1661644800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,28]],"date-time":"2022-08-28T00:00:00Z","timestamp":1661644800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2023,1]]},"DOI":"10.1007\/s00521-022-07723-2","type":"journal-article","created":{"date-parts":[[2022,8,28]],"date-time":"2022-08-28T11:02:44Z","timestamp":1661684564000},"page":"2457-2469","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Towards an efficient backbone for preserving features in speech emotion recognition: deep-shallow convolution with recurrent neural network"],"prefix":"10.1007","volume":"35","author":[{"given":"Dev Priya","family":"Goel","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kushagra","family":"Mahajan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ngoc Duy","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7527-1989","authenticated-orcid":false,"given":"Natesan","family":"Srinivasan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chee Peng","family":"Lim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,28]]},"reference":[{"issue":"02","key":"7723_CR1","doi-asserted-by":"publisher","first-page":"52","DOI":"10.38094\/jastt20291","volume":"2","author":"SMSA Abdullah","year":"2021","unstructured":"Abdullah SMSA, Ameen SYA, Sadeeq MA, Zeebaree S (2021) Multimodal emotion recognition using deep learning. J Appl Sci Technol Trends 2(02):52\u201358","journal-title":"J Appl Sci Technol Trends"},{"issue":"3\u20134","key":"7723_CR2","doi-asserted-by":"publisher","first-page":"252","DOI":"10.1016\/j.specom.2005.02.016","volume":"46","author":"T B\u00e4nziger","year":"2005","unstructured":"B\u00e4nziger T, Scherer KR (2005) The role of intonation in emotional expressions. Speech Commun 46(3\u20134):252\u2013267","journal-title":"Speech Commun"},{"issue":"3","key":"7723_CR3","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1093\/cercor\/10.3.295","volume":"10","author":"A Bechara","year":"2000","unstructured":"Bechara A, Damasio H, Damasio AR (2000) Emotion, decision making and the orbitofrontal cortex. Cereb Cortex 10(3):295\u2013307","journal-title":"Cereb Cortex"},{"issue":"10\u201311","key":"7723_CR4","doi-asserted-by":"publisher","first-page":"883","DOI":"10.1177\/0278364902021010096","volume":"21","author":"C Breazeal","year":"2002","unstructured":"Breazeal C (2002) Regulation and entrainment in human\u2013robot interaction. Int J Robot Res 21(10\u201311):883\u2013902. https:\/\/doi.org\/10.1177\/0278364902021010096","journal-title":"Int J Robot Res"},{"key":"7723_CR5","doi-asserted-by":"crossref","unstructured":"Cen L, Wu F, Yu ZL, Hu F (2016) A real-time speech emotion recognition system and its application in online learning. In: Emotions, technology, design, and learning. Elsevier, pp 27\u201346","DOI":"10.1016\/B978-0-12-801856-9.00002-5"},{"issue":"6","key":"7723_CR6","doi-asserted-by":"publisher","first-page":"1154","DOI":"10.1016\/j.dsp.2012.05.007","volume":"22","author":"L Chen","year":"2012","unstructured":"Chen L, Mao X, Xue Y, Cheng LL (2012) Speech emotion recognition: features and classification models. Digit Signal Process 22(6):1154\u20131160. https:\/\/doi.org\/10.1016\/j.dsp.2012.05.007","journal-title":"Digit Signal Process"},{"issue":"10","key":"7723_CR7","doi-asserted-by":"publisher","first-page":"1440","DOI":"10.1109\/LSP.2018.2860246","volume":"25","author":"M Chen","year":"2018","unstructured":"Chen M, He X, Yang J, Zhang H (2018) 3D convolutional recurrent neural networks with attention model for speech emotion recognition. IEEE Signal Process Lett 25(10):1440\u20131444. https:\/\/doi.org\/10.1109\/LSP.2018.2860246","journal-title":"IEEE Signal Process Lett"},{"key":"7723_CR8","doi-asserted-by":"publisher","first-page":"3515","DOI":"10.1098\/rstb.2009.0139","volume":"364","author":"R Cowie","year":"2009","unstructured":"Cowie R (2009) Perceiving emotion: towards a realistic understanding of the task. Philos Trans R Soc Lond Ser B Biol Sci 364:3515\u20133525. https:\/\/doi.org\/10.1098\/rstb.2009.0139","journal-title":"Philos Trans R Soc Lond Ser B Biol Sci"},{"issue":"1","key":"7723_CR9","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/79.911197","volume":"18","author":"R Cowie","year":"2001","unstructured":"Cowie R, Douglas-Cowie E, Tsapatsoulis N, Votsis G, Kollias S, Fellenz W, Taylor JG (2001) Emotion recognition in human\u2013computer interaction. IEEE Signal Process Mag 18(1):32\u201380","journal-title":"IEEE Signal Process Mag"},{"issue":"3","key":"7723_CR10","doi-asserted-by":"publisher","first-page":"572","DOI":"10.1016\/j.patcog.2010.09.020","volume":"44","author":"M El Ayadi","year":"2011","unstructured":"El Ayadi M, Kamel MS, Karray F (2011) Survey on speech emotion recognition: features, classification schemes, and databases. Pattern Recognit 44(3):572\u2013587. https:\/\/doi.org\/10.1016\/j.patcog.2010.09.020","journal-title":"Pattern Recognit"},{"key":"7723_CR11","doi-asserted-by":"crossref","unstructured":"ElAyadi MMH, Kamel MS, Karray F (2007) Speech emotion recognition using gaussian mixture vector autoregressive models. In: IEEE international conference on acoustics, speech and signal processing, 2007. ICASSP 2007, vol 4, pp IV-957\u2013IV-960","DOI":"10.1109\/ICASSP.2007.367230"},{"key":"7723_CR12","doi-asserted-by":"crossref","unstructured":"Giannopoulos P, Perikos I, Hatzilygeroudis I (2018) Deep learning approaches for facial emotion recognition: a case study on fer-2013. In: Advances in hybridization of intelligent methods. Springer, pp 1\u201316","DOI":"10.1007\/978-3-319-66790-4_1"},{"key":"7723_CR13","doi-asserted-by":"publisher","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: 2018 IEEE\/CVF conference on computer vision and pattern recognition, pp 7132\u20137141. https:\/\/doi.org\/10.1109\/CVPR.2018.00745","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"1","key":"7723_CR14","first-page":"235","volume":"2","author":"AB Ingale","year":"2012","unstructured":"Ingale AB, Chaudhari D (2012) Speech emotion recognition. Int J Soft Comput Eng (IJSCE) 2(1):235\u2013238","journal-title":"Int J Soft Comput Eng (IJSCE)"},{"key":"7723_CR15","doi-asserted-by":"publisher","unstructured":"Jalal M, Loweimi E, Moore R, Hain T (2019) Learning temporal clusters using capsule routing for speech emotion recognition, pp 1701\u20131705. https:\/\/doi.org\/10.21437\/Interspeech.2019-3068","DOI":"10.21437\/Interspeech.2019-3068"},{"key":"7723_CR16","doi-asserted-by":"crossref","unstructured":"Jones C, Sutherland J (2008) Acoustic emotion recognition for affective computer gaming. In: Affect and emotion in human\u2013computer interaction. Springer, pp 209\u2013219","DOI":"10.1007\/978-3-540-85099-1_18"},{"key":"7723_CR17","doi-asserted-by":"publisher","unstructured":"Lee C, Narayanan S, Pieraccini R (2002) Classifying emotions in human-machine spoken dialogs. In: Proceedings of the ICME proceedings ICME, vol 1, pp 737\u2013740. https:\/\/doi.org\/10.1109\/ICME.2002.1035887","DOI":"10.1109\/ICME.2002.1035887"},{"key":"7723_CR18","doi-asserted-by":"crossref","unstructured":"Lee J, Tashev I (2015) High-level feature representation using recurrent neural network for speech emotion recognition. Interspeech 2015. ISCA: international speech communication association","DOI":"10.21437\/Interspeech.2015-336"},{"key":"7723_CR19","doi-asserted-by":"crossref","unstructured":"Lim W, Jang D, Lee T (2016) Speech emotion recognition using convolutional and recurrent neural networks. In: 2016 Asia-Pacific signal and information processing association annual summit and conference (APSIPA), pp 1\u20134","DOI":"10.1109\/APSIPA.2016.7820699"},{"issue":"5","key":"7723_CR20","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"SR Livingstone","year":"2018","unstructured":"Livingstone SR, Russo FA (2018) The Ryerson audio-visual database of emotional speech and song (ravdess): a dynamic, multimodal set of facial and vocal expressions in North American English. PLoS ONE 13(5):e0196391","journal-title":"PLoS ONE"},{"key":"7723_CR21","doi-asserted-by":"crossref","unstructured":"Mao X, Chen L, Fu L (2009) Multi-level speech emotion recognition based on hmm and ann. In: 2009 WRI World congress on computer science and information engineering, vol 7, pp 225\u2013229","DOI":"10.1109\/CSIE.2009.113"},{"key":"7723_CR22","doi-asserted-by":"publisher","first-page":"125868","DOI":"10.1109\/ACCESS.2019.2938007","volume":"7","author":"H Meng","year":"2019","unstructured":"Meng H, Yan T, Yuan F, Wei H (2019) Speech emotion recognition from 3D log-mel spectrograms with deep learning network. IEEE Access 7:125868\u2013125881. https:\/\/doi.org\/10.1109\/ACCESS.2019.2938007","journal-title":"IEEE Access"},{"key":"7723_CR23","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1016\/S0167-6393(03)00099-2","volume":"41","author":"T Nwe","year":"2003","unstructured":"Nwe T, Foo S, De Silva L (2003) Speech emotion recognition using hidden Markov models. Speech Commun 41:603\u2013623. https:\/\/doi.org\/10.1016\/S0167-6393(03)00099-2","journal-title":"Speech Commun"},{"key":"7723_CR24","doi-asserted-by":"crossref","unstructured":"Osawa H, Orszulak J, Godfrey KM, Coughlin JF (2010) Maintaining learning motivation of older people by combining household appliance with a communication robot. In: 2010 IEEE\/RSJ international conference on intelligent robots and systems, pp 5310\u20135316","DOI":"10.1109\/IROS.2010.5648846"},{"key":"7723_CR25","unstructured":"Petrushin V (1999) Emotion in speech: recognition and application to call centers. In: Proceedings of artificial neural networks in engineering (710, 22)"},{"key":"7723_CR26","doi-asserted-by":"crossref","unstructured":"Ranganathan H, Chakraborty S, Panchanathan S (2016) Multimodal emotion recognition using deep learning architectures. In: 2016 IEEE 0(WACV), pp 1\u20139","DOI":"10.1109\/WACV.2016.7477679"},{"key":"7723_CR27","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1016\/j.visinf.2019.10.003","volume":"33","author":"M Ren","year":"2019","unstructured":"Ren M, Nie W, Liu A, Su Y (2019) Multi-modal correlated network for emotion recognition in speech. Vis Inform 33:150\u2013155","journal-title":"Vis Inform"},{"key":"7723_CR28","doi-asserted-by":"crossref","unstructured":"Rozgic V, Ananthakrishnan S, Saleem S, Kumar R, Vembu A, Prasad R (2012). Emotion recognition using acoustic and lexical features. In: 13th annual conference of the international speech communication association 2012, INTERSPEECH 2012 (1)","DOI":"10.21437\/Interspeech.2012-118"},{"key":"7723_CR29","doi-asserted-by":"publisher","unstructured":"Schuller B, Rigoll G, Lang M (2003) Hidden Markov model-based speech emotion recognition. In: 2003 International conference on multimedia and expo. ICME \u201903. Proceedings (Cat. No.03TH8698) (1, I-401). https:\/\/doi.org\/10.1109\/ICME.2003.1220939","DOI":"10.1109\/ICME.2003.1220939"},{"key":"7723_CR30","doi-asserted-by":"crossref","unstructured":"Schuller B, Rigoll G, Lang M (2004) Speech emotion recognition combining acoustic features and linguistic information in a hybrid support vector machine-belief network architecture. In: 2004 IEEE international conference on acoustics, speech, and signal processing, vol 1, pp I\u2013577","DOI":"10.1109\/ICASSP.2004.1326051"},{"key":"7723_CR31","doi-asserted-by":"crossref","unstructured":"Schuller B, Vlasenko B, Eyben F, Rigoll G, Wendemuth A (2009) Acoustic emotion recognition: a benchmark comparison of performances. In: 2009 IEEE workshop on automatic speech recognition and understanding, pp 552\u2013557","DOI":"10.1109\/ASRU.2009.5372886"},{"issue":"1","key":"7723_CR32","doi-asserted-by":"publisher","first-page":"112","DOI":"10.1049\/el.2014.3339","volume":"51","author":"P Song","year":"2015","unstructured":"Song P, Jin Y, Zha C, Zhao L (2015) Speech emotion recognition method based on hidden factor analysis. Electron Lett 51(1):112\u2013114","journal-title":"Electron Lett"},{"key":"7723_CR33","unstructured":"Sun S, Pang J, Shi J, Yi S, Ouyang W (2019) Fishnet: a versatile backbone for image, region, and pixel level prediction. arXiv preprint arXiv:1901.03495"},{"key":"7723_CR34","doi-asserted-by":"crossref","unstructured":"Tokuno S, Tsumatori G, Shono S, Takei E, Yamamoto T, Suzuki G, Shimura M (2011) Usage of emotion recognition in military health care. In: 2011 defense science research conference and expo (dsr), pp 1\u20135","DOI":"10.1109\/DSR.2011.6026823"},{"key":"7723_CR35","doi-asserted-by":"publisher","DOI":"10.3390\/brainsci9110326","author":"H Zeng","year":"2019","unstructured":"Zeng H, Wu Z, Zhang J, Yang C, Zhang H, Dai G, Kong W (2019) EEG emotion classification using an improved SincNet-based deep learning model. Brain Sci. https:\/\/doi.org\/10.3390\/brainsci9110326","journal-title":"Brain Sci"},{"key":"7723_CR36","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1016\/j.compind.2017.04.005","volume":"92","author":"Q Zhang","year":"2017","unstructured":"Zhang Q, Chen X, Zhan Q, Yang T, Xia S (2017) Respiration-based emotion recognition with deep learning. Comput Ind 92:84\u201390","journal-title":"Comput Ind"},{"key":"7723_CR37","doi-asserted-by":"crossref","unstructured":"Zhao Z, Zheng Y, Zhang Z, Wang H, Zhao Y, Li C (2018) Exploring spatio-temporal representations by integrating attention-based bidirectional-LSTM-RNNs and FCNs for speech emotion recognition interspeech","DOI":"10.21437\/Interspeech.2018-1477"},{"key":"7723_CR38","doi-asserted-by":"publisher","unstructured":"Zheng WQ, Yu JS, Zou YX (2015) An experimental study of speech emotion recognition based on deep convolutional neural networks. In: 2015 International conference on affective computing and intelligent interaction (acii), pp 827-831. https:\/\/doi.org\/10.1109\/ACII.2015.7344669","DOI":"10.1109\/ACII.2015.7344669"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-07723-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-022-07723-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-022-07723-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,16]],"date-time":"2023-01-16T05:13:14Z","timestamp":1673845994000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-022-07723-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,28]]},"references-count":38,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,1]]}},"alternative-id":["7723"],"URL":"https:\/\/doi.org\/10.1007\/s00521-022-07723-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2022,8,28]]},"assertion":[{"value":"3 November 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 August 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 August 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Yes.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Yes.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}}]}}