{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T11:17:29Z","timestamp":1784114249782,"version":"3.55.0"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10772-024-10123-7","type":"journal-article","created":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T19:02:11Z","timestamp":1720638131000},"page":"551-568","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Speech emotion recognition using the novel SwinEmoNet (Shifted Window Transformer Emotion Network)"],"prefix":"10.1007","volume":"27","author":[{"given":"R.","family":"Ramesh","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"V. B.","family":"Prahaladhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"P.","family":"Nithish","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3938-7495","authenticated-orcid":false,"given":"K.","family":"Mohanaprasad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,10]]},"reference":[{"key":"10123_CR1","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1016\/j.specom.2020.04.005","volume":"122","author":"L Abdel-Hamid","year":"2020","unstructured":"Abdel-Hamid, L. (2020). Egyptian Arabic speech emotion recognition using prosodic, spectral and wavelet features. Speech Communication, 122, 19\u201330. https:\/\/doi.org\/10.1016\/j.specom.2020.04.005","journal-title":"Speech Communication"},{"issue":"6","key":"10123_CR2","doi-asserted-by":"publisher","first-page":"2378","DOI":"10.3390\/s22062378","volume":"22","author":"A Aggarwal","year":"2022","unstructured":"Aggarwal, A., Srivastava, A., Agarwal, A., Chahal, N., Singh, D., Alnuaim, A. A., Alhadlaq, A., & Lee, H. N. (2022). Two-way feature extraction for speech emotion recognition using deep learning. Sensors, 22(6), 2378. https:\/\/doi.org\/10.3390\/s22062378","journal-title":"Sensors"},{"key":"10123_CR3","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1016\/j.specom.2019.12.001","volume":"116","author":"MB Ak\u00e7ay","year":"2020","unstructured":"Ak\u00e7ay, M. B., & O\u011fuz, K. (2020). Speech emotion recognition: Emotional models, databases, features, preprocessing methods, supporting modalities, and classifiers. Speech Communication, 116, 56\u201376. https:\/\/doi.org\/10.1016\/j.specom.2019.12.001","journal-title":"Speech Communication"},{"issue":"4","key":"10123_CR4","doi-asserted-by":"publisher","first-page":"2525","DOI":"10.1007\/s11277-023-10244-3","volume":"129","author":"MJ Al-Dujaili","year":"2023","unstructured":"Al-Dujaili, M. J., & Ebrahimi-Moghadam, A. (2023). Speech emotion recognition: A comprehensive survey. Wireless Personal Communications, 129(4), 2525\u20132561. https:\/\/doi.org\/10.1007\/s11277-023-10244-3","journal-title":"Wireless Personal Communications"},{"issue":"8","key":"10123_CR5","doi-asserted-by":"publisher","first-page":"4750","DOI":"10.3390\/app13084750","volume":"13","author":"AS Alluhaidan","year":"2023","unstructured":"Alluhaidan, A. S., Saidani, O., Jahangir, R., Nauman, M. A., & Neffati, O. S. (2023). Speech emotion recognition through hybrid features and convolutional neural network. Applied Sciences, 13(8), 4750. https:\/\/doi.org\/10.3390\/app13084750","journal-title":"Applied Sciences"},{"issue":"18","key":"10123_CR6","doi-asserted-by":"publisher","first-page":"9188","DOI":"10.3390\/app12189188","volume":"12","author":"BB Al-onazi","year":"2022","unstructured":"Al-onazi, B. B., Nauman, M. A., Jahangir, R., Malik, M. M., Alkhammash, E. H., & Elshewey, A. M. (2022). Transformer-based multilingual speech emotion recognition using data augmentation and feature fusion. Applied Sciences, 12(18), 9188. https:\/\/doi.org\/10.3390\/app12189188","journal-title":"Applied Sciences"},{"key":"10123_CR7","doi-asserted-by":"publisher","first-page":"36018","DOI":"10.1109\/access.2022.3163856","volume":"10","author":"F Andayani","year":"2022","unstructured":"Andayani, F., Theng, L. B., Tsun, M. T., & Chua, C. (2022). Hybrid LSTM-transformer model for emotion recognition from speech audio files. IEEE Access, 10, 36018\u201336027. https:\/\/doi.org\/10.1109\/access.2022.3163856","journal-title":"IEEE Access"},{"key":"10123_CR8","doi-asserted-by":"publisher","unstructured":"Bhangale, K., & Mohanaprasad, K. (2021). Speech emotion recognition using Mel frequency log spectrogram and deep convolutional neural network. In Lecture notes in electrical engineering (pp. 241\u2013250). https:\/\/doi.org\/10.1007\/978-981-16-4625-6_24","DOI":"10.1007\/978-981-16-4625-6_24"},{"issue":"4","key":"10123_CR10","doi-asserted-by":"publisher","first-page":"839","DOI":"10.3390\/electronics12040839","volume":"12","author":"K Bhangale","year":"2023","unstructured":"Bhangale, K., & Kothandaraman, M. (2023b). Speech emotion recognition based on multiple acoustic features and deep convolutional neural network. Electronics, 12(4), 839. https:\/\/doi.org\/10.3390\/electronics12040839","journal-title":"Electronics"},{"key":"10123_CR9","doi-asserted-by":"publisher","first-page":"109613","DOI":"10.1016\/j.apacoust.2023.109613","volume":"212","author":"KB Bhangale","year":"2023","unstructured":"Bhangale, K. B., & Kothandaraman, M. (2023a). Speech emotion recognition using the novel PEmoNet (Parallel Emotion Network). Applied Acoustics, 212, 109613. https:\/\/doi.org\/10.1016\/j.apacoust.2023.109613","journal-title":"Applied Acoustics"},{"key":"10123_CR11","doi-asserted-by":"publisher","unstructured":"Bhavya, S., Nayak, D. S., Dmello, R. C., Nayak, A., & Bangera, S. S. (2023, January). Machine learning applied to speech emotion analysis for depression recognition. In 2023 international conference for advancement in technology (ICONAT) (pp. 1\u20135). IEEE. https:\/\/doi.org\/10.1109\/ICONAT57137.2023.10080060","DOI":"10.1109\/ICONAT57137.2023.10080060"},{"key":"10123_CR13","doi-asserted-by":"publisher","unstructured":"Charoendee, M., Suchato, A., & Punyabukkana, P. (2017, July). Speech emotion recognition using derived features from speech segment and kernel principal component analysis. In 2017 14th international joint conference on computer science and software engineering (JCSSE) (pp. 1\u20136). IEEE. https:\/\/doi.org\/10.1109\/JCSSE.2017.8025936.","DOI":"10.1109\/JCSSE.2017.8025936"},{"key":"10123_CR14","doi-asserted-by":"crossref","unstructured":"Chen, W., Xing, X., Xu, X., Yang, J., & Pang, J. (2022, May). Key-sparse Transformer for multimodal speech emotion recognition. In 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP 2022) (pp. 6897\u20136901). IEEE.","DOI":"10.1109\/ICASSP43922.2022.9746598"},{"issue":"10","key":"10123_CR15","doi-asserted-by":"publisher","first-page":"1440","DOI":"10.1109\/lsp.2018.2860246","volume":"25","author":"M Chen","year":"2018","unstructured":"Chen, M., He, X., Yang, J., & Zhang, H. (2018). 3-D convolutional recurrent neural networks with attention model for speech emotion recognition. IEEE Signal Processing Letters, 25(10), 1440\u20131444. https:\/\/doi.org\/10.1109\/lsp.2018.2860246","journal-title":"IEEE Signal Processing Letters"},{"key":"10123_CR16","doi-asserted-by":"publisher","unstructured":"Chernyavskiy, A., Ilvovsky, D., & Nakov, P. (2021). Transformers: \u201cThe end of history\u201d for natural language processing? In Machine learning and knowledge discovery in databases: Research track: European conference, ECML PKDD 2021,  proceedings, Part III 21 (pp. 677\u2013693), Bilbao, Spain, September 13\u201317, 2021. Springer. https:\/\/doi.org\/10.48550\/arXiv.2105.00813","DOI":"10.48550\/arXiv.2105.00813"},{"issue":"15","key":"10123_CR17","doi-asserted-by":"publisher","first-page":"6972","DOI":"10.3390\/s23156972","volume":"23","author":"HC Chu","year":"2023","unstructured":"Chu, H. C., Zhang, Y. L., & Chiang, H. C. (2023). A CNN sound classification mechanism using data augmentation. Sensors, 23(15), 6972. https:\/\/doi.org\/10.3390\/s23156972","journal-title":"Sensors"},{"key":"10123_CR18","doi-asserted-by":"publisher","first-page":"221640","DOI":"10.1109\/access.2020.3043201","volume":"8","author":"MB Er","year":"2020","unstructured":"Er, M. B. (2020). A novel approach for classification of speech emotions based on deep and acoustic features. IEEE Access, 8, 221640\u2013221653. https:\/\/doi.org\/10.1109\/access.2020.3043201","journal-title":"IEEE Access"},{"issue":"1","key":"10123_CR19","doi-asserted-by":"publisher","first-page":"449","DOI":"10.1007\/s00034-022-02130-3","volume":"42","author":"MR Falahzadeh","year":"2022","unstructured":"Falahzadeh, M. R., Farokhi, F., Harimi, A., & Sabbaghi-Nadooshan, R. (2022). Deep convolutional neural network and gray wolf optimization algorithm for speech emotion recognition. Circuits, Systems, and Signal Processing, 42(1), 449\u2013492. https:\/\/doi.org\/10.1007\/s00034-022-02130-3","journal-title":"Circuits, Systems, and Signal Processing"},{"issue":"7","key":"10123_CR20","doi-asserted-by":"publisher","first-page":"4271","DOI":"10.1007\/s00034-023-02315-4","volume":"42","author":"MR Falahzadeh","year":"2023","unstructured":"Falahzadeh, M. R., Farokhi, F., Harimi, A., & Sabbaghi-Nadooshan, R. (2023). A 3D tensor representation of speech and 3D convolutional neural network for emotion recognition. Circuits, Systems, and Signal Processing, 42(7), 4271\u20134291. https:\/\/doi.org\/10.1007\/s00034-023-02315-4","journal-title":"Circuits, Systems, and Signal Processing"},{"key":"10123_CR21","doi-asserted-by":"publisher","unstructured":"Han, S., Leng, F., & Jin, Z. (2021). Speech emotion recognition with a ResNet-CNN-Transformer parallel neural network. In 2021 international conference on communications, information system and computer engineering (CISCE) (pp. 803\u2013807). IEEE. https:\/\/doi.org\/10.1109\/cisce52179.2021.9445906","DOI":"10.1109\/cisce52179.2021.9445906"},{"key":"10123_CR22","doi-asserted-by":"publisher","first-page":"109492","DOI":"10.1016\/j.apacoust.2023.109492","volume":"211","author":"C Hema","year":"2023","unstructured":"Hema, C., & Garcia Marquez, F. P. (2023). Emotional speech recognition using CNN and Deep learning techniques. Applied Acoustics, 211, 109492. https:\/\/doi.org\/10.1016\/j.apacoust.2023.109492","journal-title":"Applied Acoustics"},{"key":"10123_CR23","doi-asserted-by":"publisher","unstructured":"Ira, N. T., & Rahman, M. O. (2020, December). An efficient speech emotion recognition using ensemble method of supervised classifiers. In 2020 emerging technology in computing, communication and electronics (ETCCE) (pp. 1\u20135). IEEE. https:\/\/doi.org\/10.1109\/ETCCE51779.2020.9350913","DOI":"10.1109\/ETCCE51779.2020.9350913"},{"key":"10123_CR24","doi-asserted-by":"publisher","first-page":"101894","DOI":"10.1016\/j.bspc.2020.101894","volume":"59","author":"D Issa","year":"2020","unstructured":"Issa, D., Fatih Demirci, M., & Yazici, A. (2020). Speech emotion recognition with deep convolutional neural networks. Biomedical Signal Processing and Control, 59, 101894. https:\/\/doi.org\/10.1016\/j.bspc.2020.101894","journal-title":"Biomedical Signal Processing and Control"},{"issue":"4","key":"10123_CR25","doi-asserted-by":"publisher","first-page":"897","DOI":"10.1007\/s10772-017-9457-6","volume":"20","author":"A Jacob","year":"2017","unstructured":"Jacob, A. (2017). Modelling speech emotion recognition using logistic regression and decision trees. International Journal of Speech Technology, 20(4), 897\u2013905. https:\/\/doi.org\/10.1007\/s10772-017-9457-6","journal-title":"International Journal of Speech Technology"},{"issue":"10","key":"10123_CR26","doi-asserted-by":"publisher","first-page":"1148","DOI":"10.3844\/ajassp.2013.1148.1153","volume":"10","author":"Justin","year":"2013","unstructured":"Justin. (2013). A hybrid speech recognition system with hidden Markov model and radial basis function neural network. American Journal of Applied Sciences, 10(10), 1148\u20131153. https:\/\/doi.org\/10.3844\/ajassp.2013.1148.1153","journal-title":"American Journal of Applied Sciences"},{"issue":"1","key":"10123_CR27","doi-asserted-by":"publisher","first-page":"1523","DOI":"10.32604\/cmc.2023.028631","volume":"74","author":"S Kumar","year":"2023","unstructured":"Kumar, S., Haq, M., Jain, A., Andy Jason, C., Rao Moparthi, N., Mittal, N., & Alzamil, Z. S. (2023). Multilayer neural network based speech emotion recognition for smart assistance. Computers, Materials & Continua, 74(1), 1523\u20131540. https:\/\/doi.org\/10.32604\/cmc.2023.028631","journal-title":"Computers, Materials & Continua"},{"key":"10123_CR28","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1016\/j.procs.2015.10.020","volume":"70","author":"S Lalitha","year":"2015","unstructured":"Lalitha, S., Geyasruti, D., Narayanan, R., & Shravani, M. (2015). Emotion detection using MFCC and cepstrum features. Procedia Computer Science, 70, 29\u201335. https:\/\/doi.org\/10.1016\/j.procs.2015.10.020","journal-title":"Procedia Computer Science"},{"issue":"3","key":"10123_CR29","doi-asserted-by":"publisher","first-page":"497","DOI":"10.1007\/s10772-018-09572-8","volume":"22","author":"S Lalitha","year":"2018","unstructured":"Lalitha, S., Tripathi, S., & Gupta, D. (2018). Enhanced speech emotion detection using deep neural networks. International Journal of Speech Technology, 22(3), 497\u2013510. https:\/\/doi.org\/10.1007\/s10772-018-09572-8","journal-title":"International Journal of Speech Technology"},{"key":"10123_CR30","doi-asserted-by":"publisher","first-page":"985","DOI":"10.1109\/taslp.2021.3049898","volume":"29","author":"Z Lian","year":"2021","unstructured":"Lian, Z., Liu, B., & Tao, J. (2021). CTNet: conversational transformer network for emotion recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 29, 985\u20131000. https:\/\/doi.org\/10.1109\/taslp.2021.3049898","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"1","key":"10123_CR31","doi-asserted-by":"publisher","first-page":"012056","DOI":"10.1088\/1742-6596\/2508\/1\/012056","volume":"2508","author":"Z Liao","year":"2023","unstructured":"Liao, Z., & Shen, S. (2023). Speech emotion recognition based on Swin-transformer. Journal of Physics: Conference Series, 2508(1), 012056. https:\/\/doi.org\/10.1088\/1742-6596\/2508\/1\/012056","journal-title":"Journal of Physics: Conference Series"},{"key":"10123_CR32","doi-asserted-by":"publisher","unstructured":"Likitha, M. S., Gupta, S. R. R., Hasitha, K., & Raju, A. U. (2017, March). Speech based human emotion recognition using MFCC. In 2017 international conference on wireless communications, signal processing and networking (WiSPNET) (pp. 2257\u20132260). IEEE. https:\/\/doi.org\/10.1109\/wispnet.2017.8300161","DOI":"10.1109\/wispnet.2017.8300161"},{"key":"10123_CR33","doi-asserted-by":"publisher","unstructured":"Liu, Y., Wu, Y. H., Sun, G., Zhang, L., Chhatkuli, A., & Van Gool, L. (2021). Vision transformers with hierarchical attention. arXiv preprint arXiv:2106.03180. https:\/\/doi.org\/10.48550\/arXiv.2106.03180","DOI":"10.48550\/arXiv.2106.03180"},{"key":"10123_CR34","doi-asserted-by":"publisher","first-page":"109178","DOI":"10.1016\/j.apacoust.2022.109178","volume":"202","author":"ZT Liu","year":"2023","unstructured":"Liu, Z. T., Han, M. T., Wu, B. H., & Rehman, A. (2023). Speech emotion recognition based on convolutional neural network with attention-based bidirectional long short-term memory network and multi-task learning. Applied Acoustics, 202, 109178. https:\/\/doi.org\/10.1016\/j.apacoust.2022.109178","journal-title":"Applied Acoustics"},{"issue":"1","key":"10123_CR36","doi-asserted-by":"publisher","first-page":"327","DOI":"10.3390\/app12010327","volume":"12","author":"C Luna-Jim\u00e9nez","year":"2021","unstructured":"Luna-Jim\u00e9nez, C., Kleinlein, R., Griol, D., Callejas, Z., Montero, J. M., & Fern\u00e1ndez-Mart\u00ednez, F. (2021). A proposal for multimodal emotion recognition using aural transformers and action units on RAVDESS dataset. Applied Sciences, 12(1), 327. https:\/\/doi.org\/10.3390\/app12010327","journal-title":"Applied Sciences"},{"key":"10123_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/1687-4722-2014-21","volume":"2014","author":"HK Maganti","year":"2014","unstructured":"Maganti, H. K., & Matassoni, M. (2014). Auditory processing-based features for improving speech recognition in adverse acoustic conditions. EURASIP Journal on Audio, Speech, and Music Processing, 2014, 1\u20139. https:\/\/doi.org\/10.1186\/1687-4722-2014-21","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"key":"10123_CR38","doi-asserted-by":"publisher","first-page":"125868","DOI":"10.1109\/access.2019.2938007","volume":"7","author":"H Meng","year":"2019","unstructured":"Meng, H., Yan, T., Yuan, F., & Wei, H. (2019). Speech emotion recognition from 3D Log-Mel spectrograms with deep learning network. IEEE Access, 7, 125868\u2013125881. https:\/\/doi.org\/10.1109\/access.2019.2938007","journal-title":"IEEE Access"},{"key":"10123_CR39","doi-asserted-by":"publisher","first-page":"79861","DOI":"10.1109\/access.2020.2990405","volume":"8","author":"MS Mustaqeem","year":"2020","unstructured":"Mustaqeem, M. S., & Kwon, S. (2020). Clustering-based speech emotion recognition by incorporating learned features and deep BiLSTM. IEEE Access, 8, 79861\u201379875. https:\/\/doi.org\/10.1109\/access.2020.2990405","journal-title":"IEEE Access"},{"issue":"3","key":"10123_CR40","doi-asserted-by":"publisher","first-page":"4039","DOI":"10.32604\/cmc.2021.015070","volume":"67","author":"MS Mustaqeem","year":"2021","unstructured":"Mustaqeem, M. S., & Kwon, S. (2021). 1D-CNN: Speech emotion recognition system using a stacked network with dilated CNN features. Computers, Materials & Continua, 67(3), 4039\u20134059. https:\/\/doi.org\/10.32604\/cmc.2021.015070","journal-title":"Computers, Materials & Continua"},{"key":"10123_CR41","doi-asserted-by":"publisher","unstructured":"Omman, B., & Eldho, S. M. T. (2022, June). Speech emotion recognition using bagged support vector machines. In 2022 international conference on computing, communication, security and intelligent systems (IC3SIS) (pp. 1\u20134). IEEE. https:\/\/doi.org\/10.1109\/IC3SIS54991.2022.9885578","DOI":"10.1109\/IC3SIS54991.2022.9885578"},{"issue":"4","key":"10123_CR42","doi-asserted-by":"publisher","first-page":"26","DOI":"10.3390\/computation5020026","volume":"5","author":"M Papakostas","year":"2017","unstructured":"Papakostas, M., Spyrou, E., Giannakopoulos, T., Siantikos, G., Sgouropoulos, D., Mylonas, P., & Makedon, F. (2017). Deep visual attributes vs handcrafted audio features on multidomain speech emotion recognition. Computation, 5(4), 26. https:\/\/doi.org\/10.3390\/computation5020026","journal-title":"Computation"},{"issue":"2","key":"10123_CR43","doi-asserted-by":"publisher","first-page":"56","DOI":"10.21013\/jte.icsesd201706","volume":"7","author":"P Patel","year":"2017","unstructured":"Patel, P., Chaudhari, A. A., Pund, M. A., & Deshmukh, D. H. (2017). Speech emotion recognition system using Gaussian mixture model and improvement proposed via boosted GMM. IRA International Journal of Technology & Engineering, 7(2), 56. https:\/\/doi.org\/10.21013\/jte.icsesd201706","journal-title":"IRA International Journal of Technology & Engineering"},{"key":"10123_CR44","doi-asserted-by":"publisher","unstructured":"Pour, A. F., Asgari, M., & Hasanabadi, M. R. (2014, October). Gammatonegram based speaker identification. In 2014 4th international conference on computer and knowledge engineering (ICCKE) (pp. 52\u201355). IEEE. https:\/\/doi.org\/10.1109\/iccke.2014.6993383","DOI":"10.1109\/iccke.2014.6993383"},{"key":"10123_CR45","doi-asserted-by":"publisher","unstructured":"Saadati, M., Toroghi, R. M., & Zareian, H. (2024, February). Multi-level speaker- independent emotion recognition using complex-MFCC and Swin transformer. In 2024 20th CSI international symposium on artificial intelligence and signal processing (AISP) (pp. 1\u20134). IEEE. https:\/\/doi.org\/10.1109\/aisp61396.2024.10475274","DOI":"10.1109\/aisp61396.2024.10475274"},{"key":"10123_CR46","doi-asserted-by":"publisher","first-page":"109279","DOI":"10.1016\/j.apacoust.2023.109279","volume":"205","author":"I Shahin","year":"2023","unstructured":"Shahin, I., Alomari, O. A., Nassif, A. B., Afyouni, I., Hashem, I. A., & Elnagar, A. (2023). An efficient feature selection method for Arabic and English speech emotion recognition using Grey Wolf Optimizer. Applied Acoustics, 205, 109279. https:\/\/doi.org\/10.1016\/j.apacoust.2023.109279","journal-title":"Applied Acoustics"},{"key":"10123_CR47","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1016\/j.specom.2022.11.005","volume":"146","author":"P Singh","year":"2023","unstructured":"Singh, P., Sahidullah, M., & Saha, G. (2023). Modulation spectral features for speech emotion recognition using deep neural networks. Speech Communication, 146, 53\u201369. https:\/\/doi.org\/10.1016\/j.specom.2022.11.005","journal-title":"Speech Communication"},{"key":"10123_CR48","doi-asserted-by":"publisher","first-page":"103712","DOI":"10.1016\/j.dsp.2022.103712","volume":"130","author":"P Singh","year":"2022","unstructured":"Singh, P., Waldekar, S., Sahidullah, M., & Saha, G. (2022). Analysis of constant-Q filterbank based representations for speech emotion recognition. Digital Signal Processing, 130, 103712. https:\/\/doi.org\/10.1016\/j.dsp.2022.103712","journal-title":"Digital Signal Processing"},{"key":"10123_CR49","doi-asserted-by":"publisher","first-page":"2533","DOI":"10.1016\/j.procs.2023.01.227","volume":"218","author":"V Singh","year":"2023","unstructured":"Singh, V., & Prasad, S. (2023). Speech emotion recognition system using gender dependent convolution neural network. Procedia Computer Science, 218, 2533\u20132540. https:\/\/doi.org\/10.1016\/j.procs.2023.01.227","journal-title":"Procedia Computer Science"},{"key":"10123_CR50","doi-asserted-by":"publisher","first-page":"108637","DOI":"10.1016\/j.apacoust.2022.108637","volume":"190","author":"D Tanko","year":"2022","unstructured":"Tanko, D., Dogan, S., Burak Demir, F., Baygin, M., Engin Sahin, S., & Tuncer, T. (2022). Shoelace pattern-based speech emotion recognition of the lecturers in distance education: ShoePat23. Applied Acoustics, 190, 108637. https:\/\/doi.org\/10.1016\/j.apacoust.2022.108637","journal-title":"Applied Acoustics"},{"key":"10123_CR51","doi-asserted-by":"publisher","unstructured":"Vimal, B., Surya, M., Sridhar, V. S., & Ashok, A. (2021). MFCC based audio classification using machine learning. In 2021 12th international conference on computing communication and networking technologies (ICCCNT) (pp. 1\u20134). IEEE. https:\/\/doi.org\/10.1109\/ICCCNT51525.2021.9579881","DOI":"10.1109\/ICCCNT51525.2021.9579881"},{"key":"10123_CR52","doi-asserted-by":"publisher","unstructured":"Wang, X., Wang, M., Qi, W., Su, W., Wang, X., & Zhou, H. (2021, June). A novel end-to-end speech emotion recognition network with stacked transformer layers. In 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP 2021) (pp. 6289\u20136293). IEEE. https:\/\/doi.org\/10.1109\/icassp39728.2021.9414314","DOI":"10.1109\/icassp39728.2021.9414314"},{"key":"10123_CR53","doi-asserted-by":"publisher","unstructured":"Wang, Y., Lu, C., Lian, H., Zhao, Y., Schuller, B. W., Zong, Y., & Zheng, W. (2024, April). Speech Swin-Transformer: Exploring a hierarchical transformer with shifted windows for speech emotion recognition. In 2024 IEEE international conference on acoustics, speech and signal processing (ICASSP 2024) (pp. 11646\u201311650). IEEE. https:\/\/doi.org\/10.1109\/icassp48485.2024.10447726","DOI":"10.1109\/icassp48485.2024.10447726"},{"key":"10123_CR54","doi-asserted-by":"publisher","unstructured":"Zaman, S. R., Sadekeen, D., Alfaz, M. A., & Shahriyar, R. (2021, July). One source to detect them all: gender, age, and emotion detection from voice. In 2021 IEEE 45th annual computers, software, and applications conference (COMPSAC) (pp. 338\u2013343). IEEE. https:\/\/doi.org\/10.21203\/rs.3.rs-3502219\/v1","DOI":"10.21203\/rs.3.rs-3502219\/v1"},{"key":"10123_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, S., Liu, R., Yang, Y., Zhao, X., & Yu, J. (2022). Unsupervised domain adaptation integrating transformer and mutual information for cross-corpus speech emotion recognition. In Proceedings of the 30th ACM international conference on multimedia (pp. 120\u2013129).","DOI":"10.1145\/3503161.3548328"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10123-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10123-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10123-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,12]],"date-time":"2024-09-12T12:10:14Z","timestamp":1726143014000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10123-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,10]]},"references-count":53,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["10123"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10123-7","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,10]]},"assertion":[{"value":"23 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All the authors do not have any conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}