{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T10:27:06Z","timestamp":1784284026435,"version":"3.55.0"},"reference-count":90,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,1]],"date-time":"2024-03-01T00:00:00Z","timestamp":1709251200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s10772-024-10096-7","type":"journal-article","created":{"date-parts":[[2024,4,7]],"date-time":"2024-04-07T12:01:36Z","timestamp":1712491296000},"page":"267-285","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Automatic Speech Emotion Recognition: a Systematic Literature Review"],"prefix":"10.1007","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-3490-8596","authenticated-orcid":false,"given":"Haidy H.","family":"Mustafa","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nagy R.","family":"Darwish","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hesham A.","family":"Hefny","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,4,7]]},"reference":[{"key":"10096_CR1","unstructured":"\u201caudeering,\u201d audEERING\u00ae (2023). Retrieved May 23, 2023, from https:\/\/www.audeering.com\/research\/opensmile\/"},{"key":"10096_CR2","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1016\/j.specom.2020.04.005","volume":"122","author":"L Abdel-Hamid","year":"2020","unstructured":"Abdel-Hamid, L. (2020). Egyptian Arabic speech emotion recognition using prosodic, spectral and wavelet features. Speech Communication, 122, 19\u201330.","journal-title":"Speech Communication"},{"key":"10096_CR3","doi-asserted-by":"crossref","unstructured":"Aldeneh, Z., & Provost, E. M. (2017). Using regional saliency for speech emotion recognition. In 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), New Orleans, LA, USA.","DOI":"10.1109\/ICASSP.2017.7952655"},{"issue":"03","key":"10096_CR4","first-page":"2665","volume":"07","author":"A Al-Faham","year":"2016","unstructured":"Al-Faham, A., & Ghneim, N. (2016). Towards enhanced Arabic speech emotion recognition: Comparison between three methodologies. Asian Journal of Science and Technology, 7(3), 2665\u20132669.","journal-title":"Asian Journal of Science and Technology"},{"key":"10096_CR5","doi-asserted-by":"publisher","unstructured":"AL-Sarayreh, S., Mohamed, A., & Shaalan, K. (2023). Challenges and solutions for Arabic natural language processing in social media. In Hassanien, A.E., Zheng, D., Zhao, Z., & Fan, Z. (Eds) Business intelligence and information technology. 2022. Smart innovation, systems and technologies 358. Springer. https:\/\/doi.org\/10.1007\/978-981-99-3416-4_24","DOI":"10.1007\/978-981-99-3416-4_24"},{"issue":"1","key":"10096_CR6","volume":"1861","author":"XD An","year":"2021","unstructured":"An, X. D., & Ruan, Z. (2021). Speech emotion recognition algorithm based on deep learning algorithm fusion of temporal and spatial features. Journal of Physics: Conference Series, 1861(1), 012064.","journal-title":"Journal of Physics: Conference Series."},{"key":"10096_CR7","doi-asserted-by":"crossref","unstructured":"Anusha, R., Subhashini, P., Jyothi, D., Harshitha, P., Sushma, J., & Mukesh, N. (2021). Speech emotion recognition using machine learning. In 2021 5th international conference on trends in electronics and informatics (ICOEI), Tirunelveli, India.","DOI":"10.1109\/ICOEI51242.2021.9453028"},{"key":"10096_CR8","doi-asserted-by":"crossref","unstructured":"Aouani, H., & Ayed, Y. B. (2020). Speech emotion recognition using deep learning. In 24th international conference on knowledge-based and intelligent information & engineering, Verona, Italy.","DOI":"10.1016\/j.procs.2020.08.027"},{"key":"10096_CR9","doi-asserted-by":"publisher","first-page":"124396","DOI":"10.1109\/ACCESS.2022.3225198","volume":"10","author":"BT Atmaja","year":"2022","unstructured":"Atmaja, B. T., & Sasou, A. (2022a). Evaluating self-supervised speech representations for speech emotion recognition. IEEE Access, 10, 124396\u2013124407.","journal-title":"IEEE Access"},{"issue":"16","key":"10096_CR10","doi-asserted-by":"publisher","first-page":"5941","DOI":"10.3390\/s22165941","volume":"22","author":"BT Atmaja","year":"2022","unstructured":"Atmaja, B. T., & Sasou, A. (2022b). Effects of data augmentations on speech emotion recognition. Sensors (Basel), 22(16), 5941.","journal-title":"Sensors (Basel)"},{"issue":"17","key":"10096_CR11","doi-asserted-by":"publisher","first-page":"6369","DOI":"10.3390\/s22176369","volume":"22","author":"BT Atmaja","year":"2022","unstructured":"Atmaja, B. T., & Sasou, A. (2022c). Sentiment analysis and emotion recognition from speech using universal speech representations. Sensors, 22(17), 6369.","journal-title":"Sensors"},{"key":"10096_CR12","doi-asserted-by":"crossref","unstructured":"Atmaja, B. T., Shirai, K., & Akagi, M. (2019). Speech emotion recognition using speech feature and word embedding. In Pacific signal and information processing association annual summit and conference (APSIPA ASC), Lanzhou, China.","DOI":"10.1109\/APSIPAASC47483.2019.9023098"},{"key":"10096_CR13","doi-asserted-by":"crossref","unstructured":"Badshah, A. M., Ahmad, J., Rahim, N., & Baik, S. W. (2017). Speech emotion recognition from spectrograms with deep convolutional neural network. In 2017 international conference on platform technology and service (PlatCon), Busan, Korea (South).","DOI":"10.1109\/PlatCon.2017.7883728"},{"key":"10096_CR14","doi-asserted-by":"crossref","unstructured":"Bertero, D., & Fung, P. (2017). A first look into a convolutional neural network for speech emotion detection. In IEEE international conference on acoustics, speech and signal processing (ICASSP), New Orleans, LA, USA.","DOI":"10.1109\/ICASSP.2017.7953131"},{"issue":"13","key":"10096_CR15","doi-asserted-by":"publisher","first-page":"4653","DOI":"10.3390\/app10134653","volume":"10","author":"M Bojani\u0107","year":"2020","unstructured":"Bojani\u0107, M., Deli\u0107, V., & Karpov, A. (2020). Call redistribution for a call center based on speech emotion recognition. Applied Sciences, 10(13), 4653.","journal-title":"Applied Sciences"},{"key":"10096_CR16","doi-asserted-by":"crossref","unstructured":"Cho, J., & Kato, S. (2011). Detecting emotion from voice using selective Bayesian pairwise classifiers. In 2011 IEEE symposium on computers & informatics, Kuala Lumpur, Malaysia.","DOI":"10.1109\/ISCI.2011.5958890"},{"key":"10096_CR17","doi-asserted-by":"publisher","first-page":"32917","DOI":"10.1007\/s11042-020-09693-w","volume":"79","author":"R Dangol","year":"2020","unstructured":"Dangol, R., Alsadoon, A., Prasad, P. W. C., Seher, I., & Alsadoon, O. H. (2020). Speech emotion recognition usingconvolutional neural network and long-short term memory. Multimedia Tools and Applications, 79, 32917\u201332934.","journal-title":"Multimedia Tools and Applications"},{"issue":"1","key":"10096_CR18","doi-asserted-by":"publisher","first-page":"01","DOI":"10.14445\/22312803\/IJCTT-V52P101","volume":"52","author":"PB Dasgupta","year":"2017","unstructured":"Dasgupta, P. B. (2017). Detection and analysis of human emotions through voice and speech pattern processing. International Journal of Computer Trends and Technology (IJCTT), 52(1), 1\u20133.","journal-title":"International Journal of Computer Trends and Technology (IJCTT)"},{"issue":"1","key":"10096_CR19","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1109\/TASLP.2017.2759338","volume":"26","author":"J Deng","year":"2017","unstructured":"Deng, J., Xu, X., Zhang, Z., Fr\u00fchholz, S., & Schuller, B. (2017). Semisupervised autoencoders for speech emotion recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 26(1), 31\u201343.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"2","key":"10096_CR20","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1109\/LSP.2010.2100380","volume":"18","author":"J Dennis","year":"2011","unstructured":"Dennis, J., Dat, T. H., & Li, H. (2011). Spectrogram image feature for sound event classification in mismatched conditions. Signal Processing Letters, 18(2), 130\u2013133.","journal-title":"Signal Processing Letters"},{"key":"10096_CR21","doi-asserted-by":"crossref","unstructured":"Dissanayake, V., Zhang, H., Billinghurst, M., & Nanayakkara, S. (2020). Speech emotion recognition \u2018in the wild\u2019 using an Autoencoder. In INTERSPEECH 2020, Shanghai, China.","DOI":"10.21437\/Interspeech.2020-1356"},{"key":"10096_CR22","doi-asserted-by":"publisher","first-page":"221640","DOI":"10.1109\/ACCESS.2020.3043201","volume":"8","author":"MB Er","year":"2020","unstructured":"Er, M. B. (2020). A novel approach for classification of speech emotions based on deep and acoustic features. IEEE Access, 8, 221640\u2013221653.","journal-title":"IEEE Access"},{"key":"10096_CR23","doi-asserted-by":"crossref","unstructured":"Eskimez, S. E., Imade, K., Yang, N., Sturge-Apple, M., Duan, Z., & Heinzelman, W. (2016). Emotion classification: How does an automated system compare to Naive human coders? In IEEE international conference on acoustics, speech and signal processing (ICASSP), Shanghai, China.","DOI":"10.1109\/ICASSP.2016.7472082"},{"key":"10096_CR24","doi-asserted-by":"crossref","unstructured":"Etienne, C., Fidanza, G., Petrovskii, A., Devillers, L., & Schmauch, B. (2018). CNN+LSTM architecture for speech emotion recognition with data augmentation. In Workshop on speech, music and mind (SMM 2018), Hyderabad, India.","DOI":"10.21437\/SMM.2018-5"},{"key":"10096_CR25","doi-asserted-by":"crossref","unstructured":"Evgeniou, T. P. M. (2001). Machine learning and its applications. In Support vector machines: Theory and applications (ACAI 1999). Lecture notes in computer science, (vol. 2049). Springer.","DOI":"10.1007\/3-540-44673-7_12"},{"key":"10096_CR26","unstructured":"Feug\u00e8re, L., Doval, B., & Mifune, M.-F. (2015). Using pitch features for the characterization of intermediate vocal productions. In 5th international workshop on folk music analysis (FMA), Paris."},{"key":"10096_CR27","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2022.109133","volume":"201","author":"TML Flower","year":"2022","unstructured":"Flower, T. M. L., & Jaya, T. (2022). Speech emotion recognition using Ramanujan Fourier transform. Applied Acoustics, 201, 109133.","journal-title":"Applied Acoustics"},{"key":"10096_CR28","doi-asserted-by":"crossref","unstructured":"Gamage, K. W., Sethu, V., & Ambikairajah, E. (2017). Salience based lexical features for emotion recognition. In 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP 2017), New Orleans, LA, USA,","DOI":"10.1109\/ICASSP.2017.7953274"},{"key":"10096_CR29","doi-asserted-by":"crossref","unstructured":"Ghosh, A., Sufian, A., Sultana, F., Chakrabarti, A., & De, D. (2020). Fundamental concepts of convolutional neural network. In Recent trends and advances in artificial intelligence and Internet of Things. Intelligent systems reference. Springer.","DOI":"10.1007\/978-3-030-32644-9_36"},{"key":"10096_CR30","unstructured":"\u201cGoogle Cloud\u201d. Retrieved May 23, 2023, from https:\/\/cloud.google.com\/speech-to-text\/?utm_source=google&utm_medium=cpc&utm_campaign=emea-eg-all-en-dr-bkws-all-all-trial-e-gcp-1011340&utm_content=text-ad-none-any-DEV_c-CRE_495056377393-ADGP_Hybrid%20%7C%20BKWS%20-%20EXA%20%7C%20Txt%20~%20AI%20%26%20M"},{"key":"10096_CR31","doi-asserted-by":"crossref","unstructured":"Gupta, P., & Rajput, N. (2007). Two-stream emotion recognition for call center monitoring. In INTERSPEECH, Antwerp, Belgium.","DOI":"10.21437\/Interspeech.2007-609"},{"key":"10096_CR32","doi-asserted-by":"crossref","unstructured":"Hadjadji, I., Falek, L., Demri, L., & Teffahi, H. (2019). Emotion recognition in Arabic speech. In International conference on advanced electrical engineering (ICAEE), Algiers, Algeria.","DOI":"10.1109\/ICAEE47123.2019.9014809"},{"key":"10096_CR33","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-13-8406-6_50","volume-title":"Emotion recognition on e-learning community to improve the learning outcomes using machine learning concepts: A pilot study","author":"A Jithendran","year":"2020","unstructured":"Jithendran, A., Pranav Karthik, P., Santhosh, S., & Naren, J. (2020). Emotion recognition on e-learning community to improve the learning outcomes using machine learning concepts: A pilot study. Springer."},{"issue":"5","key":"10096_CR34","doi-asserted-by":"publisher","first-page":"1888","DOI":"10.3390\/s21051888","volume":"21","author":"J Kacur","year":"2021","unstructured":"Kacur, J., Puterka, B., Pavlovicova, J., & Oravec, M. (2021). On the speech properties and feature extraction methods in speech emotion recognition. Sensors, 21(5), 1888.","journal-title":"Sensors"},{"key":"10096_CR35","unstructured":"Kannan, V., & Rajamohan, H. R. (2019). Emotion recognition from speech, vol. 10458. arXiV:abs\/1912."},{"key":"10096_CR36","doi-asserted-by":"publisher","DOI":"10.7717\/peerj-cs.1091","volume":"8","author":"S Kanwal","year":"2022","unstructured":"Kanwal, S., Asghar, S., & Ali, H. (2022). Feature selection enhancement and feature space visualization for speech-based emotion recognition. PeerJ Computer Science, 8, e1091.","journal-title":"PeerJ Computer Science"},{"key":"10096_CR37","volume-title":"Thinkquest","author":"P Khanna","year":"2011","unstructured":"Khanna, P., & Sasikumar, M. (2011). Recognizing emotions from human speech. In S. J. Pise (Ed.), Thinkquest. Springer."},{"key":"10096_CR38","doi-asserted-by":"crossref","unstructured":"Kim, E., & Shin, J. W. (2019). DNN-based emotion recognition based on bottleneck acoustic features and lexical features. In 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), Brighton, UK.","DOI":"10.1109\/ICASSP.2019.8683077"},{"issue":"9","key":"10096_CR39","doi-asserted-by":"publisher","first-page":"2614","DOI":"10.3390\/s20092614","volume":"20","author":"E Kim","year":"2020","unstructured":"Kim, E., Song, H., & Shin, J. W. (2020a). Affective latent representation of acoustic and lexical features for emotion recognition. Sensors (Basel), 20(9), 2614.","journal-title":"Sensors (Basel)"},{"issue":"9","key":"10096_CR40","doi-asserted-by":"publisher","first-page":"2614","DOI":"10.3390\/s20092614","volume":"20","author":"E Kim","year":"2020","unstructured":"Kim, E., Song, H., & Shin, J. W. (2020b). Affective latent representation of acoustic and lexical features for emotion recognition. Sensors, 20(9), 2614.","journal-title":"Sensors"},{"issue":"4","key":"10096_CR41","first-page":"1051","volume":"45","author":"B Kitchenham","year":"2007","unstructured":"Kitchenham, B., & Charters, S. (2007). Guidelines for performing systematic literature reviews in software engineering version 2.3. Engineering, 45(4), 1051.","journal-title":"Engineering"},{"key":"10096_CR42","doi-asserted-by":"publisher","first-page":"337","DOI":"10.1007\/s10470-018-1142-4","volume":"96","author":"S Klaylat","year":"2018","unstructured":"Klaylat, S., Osman, Z., Hamandi, L., & Zantout, R. (2018). Emotion recognition in Arabic speech. Analog Integrated Circuits and Signal Processing, 96, 337\u2013351.","journal-title":"Analog Integrated Circuits and Signal Processing"},{"key":"10096_CR43","volume-title":"Autoencoders, convolutional neural networks and recurrent neural networks","author":"QV Le","year":"2015","unstructured":"Le, Q. V. (2015). Autoencoders, convolutional neural networks and recurrent neural networks. Google Inc."},{"issue":"2","key":"10096_CR44","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1109\/TSA.2004.838534","volume":"13","author":"CM Lee","year":"2005","unstructured":"Lee, C. M., & Narayanan, S. S. (2005). Toward detecting emotions in spoken dialogs. IEEE Transactions on Speech and Audio Processing, 13(2), 293\u2013303.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"10096_CR500","doi-asserted-by":"crossref","unstructured":"Lee, Y., Yoon, S., & Jung, K. (2018). Multimodal speech emotion recognition using audio and text. In 2018 IEEE spoken language technology workshop (SLT), Athens, Greece.","DOI":"10.1109\/SLT.2018.8639583"},{"key":"10096_CR45","doi-asserted-by":"crossref","unstructured":"Li, B., Dimitriadis, D., & Stolcke, A. (2019). Acoustic and lexical sentiment analysis for customer service calls. In IEEE international conference on acoustics, speech and signal processing (ICASSP), Brighton, UK.","DOI":"10.1109\/ICASSP.2019.8683679"},{"issue":"21","key":"10096_CR46","doi-asserted-by":"publisher","first-page":"8152","DOI":"10.3390\/s22218152","volume":"22","author":"GM Li","year":"2022","unstructured":"Li, G. M., Liu, N., & Zhang, J.-A. (2022). Speech emotion recognition based on modified relief. Sensors (Basel), 22(21), 8152.","journal-title":"Sensors (Basel)"},{"key":"10096_CR600","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhang, Y.-T., Ng, G. W., Leau, Y.-B., & Yan, H. (2023). A deep learning method using gender-specific features for emotion recognition: A deep learning method using gender-specific features for emotion recognition. Sensors, 23(3), 1355\u20131356.","DOI":"10.3390\/s23031355"},{"key":"10096_CR47","unstructured":"\u201clibrosa\u201d. Retrieved May 23, 2023, from https:\/\/librosa.org\/doc\/latest\/index.html"},{"key":"10096_CR48","doi-asserted-by":"crossref","unstructured":"Lieskovska, E., Jakubec, M., & Jarina, R. (2022). RNN with improved temporal modeling for speech emotion recognition. In 2022 32nd international conference radioelektronika (RADIOELEKTRONIKA), Kosice, Slovakia.","DOI":"10.1109\/RADIOELEKTRONIKA54537.2022.9764901"},{"key":"10096_CR49","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1007\/s10772-021-09955-4","volume":"25","author":"M Liu","year":"2022","unstructured":"Liu, M. (2022). English speech emotion recognition method based on speech recognition. International Journal of Speech Technology, 25, 391\u2013398.","journal-title":"International Journal of Speech Technology"},{"key":"10096_CR50","doi-asserted-by":"publisher","first-page":"95925","DOI":"10.1109\/ACCESS.2021.3094355","volume":"9","author":"N Liu","year":"2021","unstructured":"Liu, N., Zhang, B., Liu, B., Shi, J., Yang, L., Li, Z., & Zhu, J. (2021). Transfer subspace learning for unsupervised cross-corpus speech emotion recognition. IEEE Access, 9, 95925\u201395937.","journal-title":"IEEE Access"},{"issue":"4","key":"10096_CR51","doi-asserted-by":"publisher","DOI":"10.1088\/1742-6596\/1748\/4\/042008","volume":"1748","author":"X Lun","year":"2021","unstructured":"Lun, X., Wang, F., & Yu, Z. (2021). Human speech emotion recognition via feature selection and analyzing. Journal of Physics Conference Series, 1748(4), 042008.","journal-title":"Journal of Physics Conference Series"},{"key":"10096_CR700","doi-asserted-by":"crossref","unstructured":"Maghilnan, S., & Kumar, M. R. (2017). Sentiment analysis on speaker specific speech data. In 2017 international conference on intelligent computing and control (I2C2), Coimbatore, India.","DOI":"10.1109\/I2C2.2017.8321795"},{"issue":"1","key":"10096_CR52","first-page":"38","volume":"79","author":"SA Majeed","year":"2015","unstructured":"Majeed, S. A., Husain, H., Samad, S. A., & Idbeaa, T. F. (2015). Mel frequency cepstral coefficients (MFCC) feature extraction enhancement in theapplication of speech recognition: A comparison study. Journal of Theoretical and Applied Information Technology, 79(1), 38.","journal-title":"Journal of Theoretical and Applied Information Technology"},{"key":"10096_CR53","unstructured":"\u201cMathWorks\u201d. Retrieved May 23, 2023, from https:\/\/www.mathworks.com\/products\/matlab.html"},{"key":"10096_CR54","first-page":"184","volume":"8","author":"M Meddeb","year":"2016","unstructured":"Meddeb, M., Karray, H., & Alimi, A. M. (2016). Automated extraction of features from arabic emotional speech corpus. International Journal of Computer Information Systems and Industrial Management Applications, 8, 184\u2013194.","journal-title":"International Journal of Computer Information Systems and Industrial Management Applications."},{"key":"10096_CR55","doi-asserted-by":"crossref","unstructured":"Mefiah, A., Alotaibi, Y. A., & Selouani, S.-A. (2015). Arabic speaker emotion classification using rhythm metrics and neural networks. In 2015 23rd European signal processing conference (EUSIPCO), Nice, France.","DOI":"10.1109\/EUSIPCO.2015.7362619"},{"key":"10096_CR56","doi-asserted-by":"crossref","unstructured":"Meftah, A., Selouani, S.-A., & Alotaibi, Y. A. (2015). Preliminary Arabic speech emotion classification. In 2014 IEEE international symposium on signal processing and information technology (ISSPIT), Noida, India.","DOI":"10.1109\/ISSPIT.2014.7300584"},{"key":"10096_CR57","doi-asserted-by":"crossref","unstructured":"Meftah, A., Qamhan, M., Alotaibi, Y. A., & Zakariah, M. (2020). Arabic speech emotion recognition using KNN and KSU emotions corpus. International Journal of Simulation -- Systems, Science & Technology, 21(2), 1\u20135.","DOI":"10.5013\/IJSSST.a.21.02.21"},{"key":"10096_CR58","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1016\/j.specom.2019.12.001","volume":"116","author":"B Mehmet","year":"2020","unstructured":"Mehmet, B., & Kaya, O. (2020). Speech emotion recognition: Emotional models, databases, features, preprocessing methods, supporting modalities, and classifiers. Speech Communication, 116, 56\u201376.","journal-title":"Speech Communication"},{"key":"10096_CR59","doi-asserted-by":"publisher","DOI":"10.37200\/IJPR\/V24I8\/PR280260","author":"H Murugan","year":"2020","unstructured":"Murugan, H. (2020). Speech emotion recognition using CNN. International Journal of Psychosocial Rehabilitation. https:\/\/doi.org\/10.37200\/IJPR\/V24I8\/PR280260","journal-title":"International Journal of Psychosocial Rehabilitation"},{"key":"10096_CR800","doi-asserted-by":"crossref","unstructured":"Naziya, S., & Ratnadeep, R. D. (2016). Speech recognition system\u2014a review. IOSR Journal of Computer Engineering, 18(4), 1\u20139.","DOI":"10.9790\/0661-1804020109"},{"key":"10096_CR60","doi-asserted-by":"crossref","unstructured":"Pengfei, X., Houpan, Z., & Weidong, Z. (2020). PAD 3-D speech emotion recognition based on feature fusion. Journal of Physics Conference Series 1616, 012106.","DOI":"10.1088\/1742-6596\/1616\/1\/012106"},{"key":"10096_CR61","doi-asserted-by":"publisher","first-page":"5311","DOI":"10.3390\/s22145311","volume":"22","author":"M P\u0142aza","year":"2022","unstructured":"P\u0142aza, M., Trusz, S., K\u0119czkowska, J., Boksa, E., Sadowski, S., & Koruba, Z. (2022). Machine learning algorithms for detection and classifications of emotions in contact center applications. Sensors, 22, 5311.","journal-title":"Sensors"},{"key":"10096_CR62","unstructured":"\u201cpython\u201d. Retrieved May 23, 2023, from https:\/\/www.python.org\/"},{"issue":"5","key":"10096_CR63","first-page":"422","volume":"5","author":"A Rawat","year":"2015","unstructured":"Rawat, A., & Mishra, P. K. (2015). Emotion recognition through speech using neural network. International Journal of Advanced Research in Computer Science and Software Engineering, 5(5), 422\u2013428.","journal-title":"International Journal of Advanced Research in Computer Science and Software Engineering"},{"key":"10096_CR64","doi-asserted-by":"crossref","unstructured":"Sahu, S., Mitra, V., Seneviratne, S., & Espy-Wilson, C. (2019). Multi-modal learning for speech emotion recognition: An analysis and comparison of ASR outputs with ground truth transcription. In Proceedings of Interspeech (pp. 3302\u20133306).","DOI":"10.21437\/Interspeech.2019-1149"},{"key":"10096_CR65","doi-asserted-by":"crossref","unstructured":"Sato, S., Kimura, T., Horiuchi, Y., & Nishida, M. (2008). A method for automatically estimating F0 model parameters and a speech re-synthesis tool using F0 model and STRAIGHT. In INTERSPEECH 2008, 9th annual conference of the international speech communication association, Brisbane, Australia.","DOI":"10.21437\/Interspeech.2008-162"},{"key":"10096_CR66","doi-asserted-by":"crossref","unstructured":"Schuller, B., Rigoll, G., &. Manfred, L. (2004). Speech emotion recognition combining acoustic features and linguistic information in a hybrid support vector machine-belief network architecture. In IEEE international conference on acoustics, speech, and signal processing (ICASSP), Montreal, QC, Canada.","DOI":"10.1109\/ICASSP.2004.1326051"},{"key":"10096_CR67","doi-asserted-by":"crossref","unstructured":"Seknedy, M. E., & Fawzi, S. (2021). Speech emotion recognition system for human interaction applications. In 10th international conference on intelligent computing and information systems (ICICIS), Cairo, Egypt.","DOI":"10.1109\/ICICIS52592.2021.9694246"},{"issue":"1","key":"10096_CR68","first-page":"311","volume":"8","author":"M Selvara","year":"2016","unstructured":"Selvara, M., Bhuvana, R., & Padmaja, S. (2016). Human speech emotion recognition. International Journal of Engineering and Technology (IJET), 8(1), 311\u2013323.","journal-title":"International Journal of Engineering and Technology (IJET)"},{"key":"10096_CR69","doi-asserted-by":"publisher","DOI":"10.1016\/j.dcan.2022.10.018","author":"P Shixin","year":"2022","unstructured":"Shixin, P., Kai, C., Tian, T., & Jingying, C. (2022). An autoencoder-based feature level fusion for speech emotion recognition. Digital Communications and Networks. https:\/\/doi.org\/10.1016\/j.dcan.2022.10.018","journal-title":"Digital Communications and Networks"},{"key":"10096_CR70","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1016\/j.neucom.2022.04.028","volume":"492","author":"YB Singh","year":"2022","unstructured":"Singh, Y. B., & Goel, S. (2022). A systematic literature review of speech emotion recognition approaches. Neurocomputing, 492, 245\u2013263.","journal-title":"Neurocomputing"},{"key":"10096_CR71","unstructured":"Srivastava, B. M. L., Kajarekar, S., & Murthy, H. A. (2019). Challenges in automatic transcription of real-world phone calls. In Proceedings of Interspeech, Graz, Austria."},{"key":"10096_CR72","doi-asserted-by":"publisher","first-page":"1075624","DOI":"10.3389\/fpsyg.2022.1075624","volume":"13","author":"C Sun","year":"2023","unstructured":"Sun, C., Li, H., & Ma, L. (2023). Speech emotion recognition based on improved masking EMD and convolutional recurrent neural network. Frontiers in Psychology, 13, 1075624.","journal-title":"Frontiers in Psychology"},{"key":"10096_CR73","first-page":"1","volume":"2","author":"L Sun","year":"2019","unstructured":"Sun, L., Fu, S., & Wang, F. (2019). Decision tree SVM model with Fisher feature selection for speech emotion recognition. EURASIP Journal on Audio, Speech, and Music Processing, 2, 1\u201314.","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"key":"10096_CR74","doi-asserted-by":"crossref","unstructured":"Tacconi, D., Mayora, O., Lukowicz, P., Arnrich, B., Setz, C., Troster, G., & Haring, C. (2008). Activity and emotion recognition to support early diagnosis of psychiatric diseases. In International conference on pervasive computing technologies for healthcare.","DOI":"10.4108\/ICST.PERVASIVEHEALTH2008.2511"},{"key":"10096_CR75","unstructured":"\u201cThe University of Waikato\u201d. Retrieved May 23, 2023, from https:\/\/www.cs.waikato.ac.nz\/ml\/weka\/"},{"key":"10096_CR76","doi-asserted-by":"crossref","unstructured":"Trigeorgis, G., Ringeval, F., Brueckner, R., Marchi, E., Nicolaou, M. A., Schuller, B., & Zafeiriou, S. (2016). Adieu features? End-to-end speech emotion recognition using a deep convolutional recurrent network. In IEEE international conference on acoustics, speech and signal processing (ICASSP), Shanghai, China.","DOI":"10.1109\/ICASSP.2016.7472669"},{"key":"10096_CR77","doi-asserted-by":"crossref","unstructured":"Wani, T. M., Gunawan, T. S., Qadri, S. A. A., Mansor, H., Kartiwi, M., & Ismail, N. (2020). Speech emotion recognition using convolution neural networks and deep stride convolutional neural networks. In 6th international conference on wireless and telematics (ICWT), Yogyakarta, Indonesia.","DOI":"10.1109\/ICWT50448.2020.9243622"},{"key":"10096_CR78","unstructured":"\u201cWavePad Audio Editing Software\u201d. Retrieved May 23, 2023, from https:\/\/www.nch.com.au\/wavepad\/index.html"},{"key":"10096_CR79","doi-asserted-by":"publisher","first-page":"27","DOI":"10.1007\/s10772-016-9364-2","volume":"20","author":"N Yang","year":"2017","unstructured":"Yang, N., Yuan, J., Zhou, Y., Demirkol, I., Duan, Z., Heinzelman, W., & Sturge-Apple, M. (2017). Enhanced multiclass SVM with thresholding fusion for speech-based emotion classification. International Journal of Speech Technology, 20, 27\u201341.","journal-title":"International Journal of Speech Technology"},{"key":"10096_CR80","doi-asserted-by":"crossref","unstructured":"Yazdani, A., Simchi, H., & Shekofteh, Y. (2021). Emotion recognition in persian speech using deep neural networks. In 11th international conference on computer engineering and knowledge (ICCKE), Mashhad, Iran.","DOI":"10.1109\/ICCKE54056.2021.9721504"},{"issue":"5","key":"10096_CR81","doi-asserted-by":"publisher","first-page":"713","DOI":"10.3390\/electronics9050713","volume":"9","author":"Y Yu","year":"2020","unstructured":"Yu, Y., & Kim, Y.-J. (2020). Attention-LSTM-attention model for speech emotion recognition and analysis of IEMOCAP database. Electronics, 9(5), 713.","journal-title":"Electronics"},{"issue":"5","key":"10096_CR83","volume":"9","author":"Y Zhang","year":"2022","unstructured":"Zhang, Y., & Srivastava, G. (2022). Speech emotion recognition method in educational scene based on machine learning. EAI Endorsed Transactions on Scalable Information Systems, 9(5), e9.","journal-title":"EAI Endorsed Transactions on Scalable Information Systems"},{"key":"10096_CR84","doi-asserted-by":"publisher","first-page":"312","DOI":"10.1016\/j.bspc.2018.08.035","volume":"47","author":"J Zhao","year":"2018","unstructured":"Zhao, J., Mao, X., & Chen, L. (2018). Speech emotion recognition using deep 1D & 2D CNN LSTM networks. Biomedical Signal Processing and Control, 47, 312\u2013323.","journal-title":"Biomedical Signal Processing and Control"},{"issue":"1","key":"10096_CR85","doi-asserted-by":"publisher","first-page":"205","DOI":"10.3390\/app10010205","volume":"10","author":"C Zheng","year":"2020","unstructured":"Zheng, C., Wang, C., & Jia, N. (2020). An ensemble model for multi-level speech emotion recognition. Applied Sciences, 10(1), 205\u2013224.","journal-title":"Applied Sciences"},{"key":"10096_CR86","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2020.103775","volume":"94","author":"M ZiaUddin","year":"2020","unstructured":"ZiaUddin, M., & Nilsson, E. G. (2020). Emotion recognition using speech and neural structured learning to facilitate edge intelligence. Engineering Applications of Artificial Intelligence, 94, 103775.","journal-title":"Engineering Applications of Artificial Intelligence"},{"issue":"5","key":"10096_CR87","doi-asserted-by":"publisher","first-page":"1065","DOI":"10.3233\/IDA-194747","volume":"24","author":"K Zvarevashe","year":"2020","unstructured":"Zvarevashe, K., & Olugbara, O. O. (2020). Recognition of speech emotion using custom 2D-convolution neural network deep learning algorithm. Intelligent Data Analysis, 24(5), 1065\u20131086.","journal-title":"Intelligent Data Analysis"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10096-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10096-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10096-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T15:16:44Z","timestamp":1715613404000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10096-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3]]},"references-count":90,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["10096"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10096-7","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3]]},"assertion":[{"value":"19 December 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 February 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 April 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"None of the authors have any competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}