{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:14:01Z","timestamp":1781586841865,"version":"3.54.5"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s11760-024-03406-8","type":"journal-article","created":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T18:02:10Z","timestamp":1720634530000},"page":"7445-7454","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Enhanced speech emotion recognition using averaged valence arousal dominance mapping and deep neural networks"],"prefix":"10.1007","volume":"18","author":[{"given":"Davit","family":"Rizhinashvili","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdallah Hussein","family":"Sham","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gholamreza","family":"Anbarjafari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,10]]},"reference":[{"issue":"1","key":"3406_CR1","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1109\/79.911197","volume":"18","author":"R Cowie","year":"2001","unstructured":"Cowie, R., Douglas-Cowie, E., Tsapatsoulis, N., Votsis, G., Kollias, S., Fellenz, W., Taylor, J.G.: Emotion recognition in human-computer interaction. IEEE Signal Process. Mag. 18(1), 32\u201380 (2001)","journal-title":"IEEE Signal Process. Mag."},{"key":"3406_CR2","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Sun, Y., Zhang, J., Yan, Y.: Speech emotion recognition using both spectral and prosodic features. In: International Conference on Information Engineering and Computer Science. IEEE 2009, 1\u20134 (2009)","DOI":"10.1109\/ICIECS.2009.5362730"},{"key":"3406_CR3","doi-asserted-by":"crossref","unstructured":"Schneider, S., Baevski, A., Collobert, R., Auli, M.: wav2vec: Unsupervised pre-training for speech recognition, arXiv preprint arXiv:1904.05862, 2019","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"3406_CR4","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1007\/BF00993070","volume":"8","author":"RB Hupka","year":"1984","unstructured":"Hupka, R.B.: Jealousy: Compound emotion or label for a particular situation? Motiv. Emot. 8, 141\u2013155 (1984)","journal-title":"Motiv. Emot."},{"key":"3406_CR5","doi-asserted-by":"publisher","first-page":"2159","DOI":"10.1007\/s11042-015-3119-y","volume":"76","author":"GK Verma","year":"2017","unstructured":"Verma, G.K., Tiwary, U.S.: Affect representation and recognition in 3d continuous valence-arousal-dominance space. Multimed. Tools Appl. 76, 2159\u20132183 (2017)","journal-title":"Multimed. Tools Appl."},{"key":"3406_CR6","doi-asserted-by":"publisher","first-page":"1191","DOI":"10.3758\/s13428-012-0314-x","volume":"45","author":"AB Warriner","year":"2013","unstructured":"Warriner, A.B., Kuperman, V., Brysbaert, M.: Norms of valence, arousal, and dominance for 13,915 English lemmas. Behav. Res. Methods 45, 1191\u20131207 (2013)","journal-title":"Behav. Res. Methods"},{"key":"3406_CR7","first-page":"449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. Adv. Neural Inform. Process. Syst. 33, 449\u2013460 (2020)","journal-title":"Adv. Neural Inform. Process. Syst."},{"issue":"3","key":"3406_CR8","doi-asserted-by":"publisher","first-page":"412","DOI":"10.1037\/0033-2909.97.3.412","volume":"97","author":"RW Frick","year":"1985","unstructured":"Frick, R.W.: Communicating emotion: the role of prosodic features. Psychol. Bull. 97(3), 412 (1985)","journal-title":"Psychol. Bull."},{"key":"3406_CR9","unstructured":"Alter, K., Rank, E., Kotz, S.A., Pfeifer, E., Besson, M., Friederici, A.D., Matiasek, J.: On the Relations of Semantic and Acoustic Properties of Emotions. (1999)"},{"key":"3406_CR10","doi-asserted-by":"publisher","first-page":"347","DOI":"10.1023\/A:1023237014909","volume":"28","author":"C Sobin","year":"1999","unstructured":"Sobin, C., Alpert, M.: Emotion in speech: the acoustic attributes of fear, anger, sadness, and joy. J. Psycholinguist. Res. 28, 347\u2013365 (1999)","journal-title":"J. Psycholinguist. Res."},{"issue":"3","key":"3406_CR11","doi-asserted-by":"publisher","first-page":"614","DOI":"10.1037\/0022-3514.70.3.614","volume":"70","author":"R Banse","year":"1996","unstructured":"Banse, R., Scherer, K.R.: Acoustic profiles in vocal emotion expression. J. Pers. Soc. Psychol. 70(3), 614 (1996)","journal-title":"J. Pers. Soc. Psychol."},{"key":"3406_CR12","doi-asserted-by":"publisher","first-page":"155","DOI":"10.1007\/s10462-012-9368-5","volume":"43","author":"C Anagnostopoulos","year":"2015","unstructured":"Anagnostopoulos, C., Iliou, T., Giannoukos, I.: Features and classifiers for emotion recognition from speech: a survey from 2000 to 2011. Artif. Intell. Rev. 43, 155\u2013177 (2015)","journal-title":"Artif. Intell. Rev."},{"key":"3406_CR13","doi-asserted-by":"crossref","unstructured":"Khalil, R.A., Jones, E., Babar, M.I., Jan, T., Zafar, M.H., Alhussain, T.: \u201cSpeech emotion recognition using deep learning techniques: A review,\u201d IEEE Access, vol.\u00a07, pp. 117\u00a0327\u2013117\u00a0345, 2019","DOI":"10.1109\/ACCESS.2019.2936124"},{"key":"3406_CR14","doi-asserted-by":"crossref","unstructured":"Bharti, D., Kukana, P.: \u201cA hybrid machine learning model for emotion recognition from speech signals,\u201d 2020 International Conference on Smart Electronics and Communication (ICOSEC), pp. 491\u2013496, 2020","DOI":"10.1109\/ICOSEC49089.2020.9215376"},{"key":"3406_CR15","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/s10772-017-9396-2","volume":"20","author":"F Noroozi","year":"2017","unstructured":"Noroozi, F., Sapinski, T., Kaminska, D., Anbarjafari, G.: Vocal-based emotion recognition using random forests and decision tree. Int. J. Speech Technol. 20, 239\u2013246 (2017)","journal-title":"Int. J. Speech Technol."},{"key":"3406_CR16","unstructured":"Anand, N., Verma, P.: Convoluted Feelings Convolutional and Recurrent Nets for Detecting Emotion from Audio Data, (2015)"},{"issue":"10","key":"3406_CR17","doi-asserted-by":"publisher","first-page":"1440","DOI":"10.1109\/LSP.2018.2860246","volume":"25","author":"M Chen","year":"2018","unstructured":"Chen, M., He, X., Yang, J., Zhang, H.: 3-d convolutional recurrent neural networks with attention model for speech emotion recognition. IEEE Signal Process. Lett. 25(10), 1440\u20131444 (2018)","journal-title":"IEEE Signal Process. Lett."},{"key":"3406_CR18","doi-asserted-by":"publisher","first-page":"90368","DOI":"10.1109\/ACCESS.2019.2927384","volume":"7","author":"P Jiang","year":"2019","unstructured":"Jiang, P., Fu, H., Tao, H., Lei, P., Zhao, L.: Parallelized convolutional recurrent neural network with spectral features for speech emotion recognition. IEEE Access 7, 90368\u201390377 (2019)","journal-title":"IEEE Access"},{"issue":"6","key":"3406_CR19","doi-asserted-by":"publisher","first-page":"550","DOI":"10.1049\/iet-bmt.2018.5074","volume":"7","author":"P Tertychnyi","year":"2018","unstructured":"Tertychnyi, P., Ozcinar, C., Anbarjafari, G.: Low-quality fingerprint classification using deep neural network. IET Biometrics 7(6), 550\u2013556 (2018)","journal-title":"IET Biometrics"},{"key":"3406_CR20","doi-asserted-by":"publisher","first-page":"125868","DOI":"10.1109\/ACCESS.2019.2938007","volume":"7","author":"H Meng","year":"2019","unstructured":"Meng, H., Yan, T., Yuan, F., Wei, F.: Speech emotion recognition from 3d log-mel spectrograms with deep learning network. IEEE Access 7, 125868\u2013125881 (2019)","journal-title":"IEEE Access"},{"key":"3406_CR21","doi-asserted-by":"publisher","first-page":"79861","DOI":"10.1109\/ACCESS.2020.2990405","volume":"8","author":"M Sajjad","year":"2020","unstructured":"Sajjad, M., Kwon, S., et al.: Clustering-based speech emotion recognition by incorporating learned features and deep bilstm. IEEE Access 8, 79861\u201379875 (2020)","journal-title":"IEEE Access"},{"key":"3406_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2020.101894","volume":"59","author":"D Issa","year":"2020","unstructured":"Issa, D., Demirci, M.F., Yazici, A.: Speech emotion recognition with deep convolutional neural networks. Biomed. Signal Process. Control 59, 101894 (2020)","journal-title":"Biomed. Signal Process. Control"},{"key":"3406_CR23","doi-asserted-by":"publisher","unstructured":"Livingstone, S.R., Russo, F.A.: The Ryerson Audio-Visual Database of Emotional Speech and Song (RAVDESS), (2018). [Online]. Available: https:\/\/doi.org\/10.5281\/zenodo.1188976","DOI":"10.5281\/zenodo.1188976"},{"key":"3406_CR24","doi-asserted-by":"crossref","unstructured":"Burkhardt, F., Paeschke, A., Rolfes, M., Sendlmeier, W.F., Weiss B. et\u00a0al.: A database of German emotional speech. In: Interspeech, vol.\u00a05, pp. 1517\u20131520 (2005)","DOI":"10.21437\/Interspeech.2005-446"},{"issue":"1","key":"3406_CR25","doi-asserted-by":"publisher","first-page":"778","DOI":"10.1038\/s41597-022-01855-9","volume":"9","author":"RY Paccotacya-Yanque","year":"2022","unstructured":"Paccotacya-Yanque, R.Y., Huanca-Anquise, C.A., Escalante-Calcina, J., Ramos-Lov\u00f3n, W.R., Cuno-Parari, \u00c1.E.: A speech corpus of Quechua Collao for automatic dimensional emotion recognition. Sci. Data 9(1), 778 (2022)","journal-title":"Sci. Data"},{"issue":"10","key":"3406_CR26","doi-asserted-by":"publisher","first-page":"1594","DOI":"10.3390\/electronics11101594","volume":"11","author":"D Rizhinashvili","year":"2022","unstructured":"Rizhinashvili, D., Sham, A.H., Anbarjafari, G.: Gender neutralisation for unbiased speech synthesising. Electronics 11(10), 1594 (2022)","journal-title":"Electronics"},{"key":"3406_CR27","doi-asserted-by":"crossref","unstructured":"Pepino, L., Riera, P., Ferrer, L.: Emotion recognition from speech using wav2vec 2.0 embeddings, arXiv preprint arXiv:2104.03502, (2021)","DOI":"10.21437\/Interspeech.2021-703"},{"key":"3406_CR28","doi-asserted-by":"crossref","unstructured":"Neumann, M., Vu, N.T.: Investigations on audiovisual emotion recognition in noisy conditions. In: IEEE Spoken Language Technology Workshop (SLT). IEEE vol. 2021, pp. 358\u2013364 (2021)","DOI":"10.1109\/SLT48900.2021.9383588"},{"key":"3406_CR29","doi-asserted-by":"crossref","unstructured":"Yannakakis, G.N., Cowie, R., Busso, C.: The ordinal nature of emotions. In: 2017 Seventh International Conference on Affective Computing and Intelligent Interaction (ACII). IEEE, pp. 248\u2013255 (2017)","DOI":"10.1109\/ACII.2017.8273608"},{"issue":"1","key":"3406_CR30","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1177\/1529100619850176","volume":"20","author":"A Cowen","year":"2019","unstructured":"Cowen, A., Sauter, D., Tracy, J.L., Keltner, D.: Mapping the passions: toward a high-dimensional taxonomy of emotional experience and expression. Psychol. Sci. Public Interest 20(1), 69\u201390 (2019)","journal-title":"Psychol. Sci. Public Interest"},{"key":"3406_CR31","unstructured":"Buechel, S., Hahn, U.: Representation mapping: a novel approach to generate high-quality multi-lingual emotion lexicons, arXiv preprint arXiv:1807.00775, 2018"},{"key":"3406_CR32","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2023.104894","volume":"85","author":"D Nandini","year":"2023","unstructured":"Nandini, D., Yadav, J., Rani, A., Singh, V.: Design of subject independent 3d VAD emotion detection system using EEG signals and machine learning algorithms. Biomed. Signal Process. Control 85, 104894 (2023)","journal-title":"Biomed. Signal Process. Control"},{"key":"3406_CR33","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1007\/978-3-030-96993-6_8","volume-title":"Biologically Inspired Cognitive Architectures 2021: Proceedings of the 12th Annual Meeting of the BICA Society","author":"A Dolidze","year":"2022","unstructured":"Dolidze, A., Morozevich, M., Pak, N.: Mapping Speech Intonations to the VAD Model of Emotions. In: Klimov, V.V., Kelley, D.J. (eds.) Biologically Inspired Cognitive Architectures 2021: Proceedings of the 12th Annual Meeting of the BICA Society, pp. 89\u201395. Springer International Publishing, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-030-96993-6_8"},{"key":"3406_CR34","unstructured":"Park, S., Kim, J., Ye, S., Jeon, J., Y. H., Park, Oh, A.: Dimensional emotion detection from categorical emotion, arXiv preprint arXiv:1911.02499, (2019)"},{"key":"3406_CR35","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1142\/9789812775320_0021","volume-title":"Handbook of Pattern Recognition and Computer Vision","author":"N Sebe","year":"2011","unstructured":"Sebe, N., Cohen, I., Huang, T.S.: Multimodal Emotion Recognition. In: Chen, C.H., Wang, P.S.P. (eds.) Handbook of Pattern Recognition and Computer Vision, pp. 387\u2013409. World Scientific, Singapore (2011). https:\/\/doi.org\/10.1142\/9789812775320_0021"},{"key":"3406_CR36","doi-asserted-by":"crossref","unstructured":"S.\u00a0Haq and P.\u00a0J. Jackson, Multimodal emotion recognition. In: Machine audition: principles, algorithms and systems. IGI Global, pp. 398\u2013423 (2011)","DOI":"10.4018\/978-1-61520-919-4.ch017"},{"key":"3406_CR37","doi-asserted-by":"crossref","unstructured":"Gorbova, J., Lusi, I., Litvin, A., Anbarjafari, G.: Automated screening of job candidate based on multimodal video processing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 29\u201335 (2017)","DOI":"10.1109\/CVPRW.2017.214"},{"key":"3406_CR38","doi-asserted-by":"crossref","unstructured":"Hook, J., Noroozi, F., Toygar, O., Anbarjafari, G.: Automatic speech based emotion recognition using paralinguistics features. In: Bulletin of the Polish Academy of Sciences. Technical Sciences, vol.\u00a067, no.\u00a03, (2019)","DOI":"10.24425\/bpasts.2019.129647"},{"issue":"02","key":"3406_CR39","first-page":"52","volume":"2","author":"SMSA Abdullah","year":"2021","unstructured":"Abdullah, S.M.S.A., Ameen, S.Y.A., Sadeeq, M.A., Zeebaree, S.: Multimodal emotion recognition using deep learning. J. Appl. Sci. Technol. Trends 2(02), 52\u201358 (2021)","journal-title":"J. Appl. Sci. Technol. Trends"},{"key":"3406_CR40","doi-asserted-by":"crossref","unstructured":"Jaitly, N., Hinton, G.: Learning a better representation of speech soundwaves using restricted boltzmann machines. In: 2011 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp. 5884\u20135887 (2011)","DOI":"10.1109\/ICASSP.2011.5947700"},{"key":"3406_CR41","doi-asserted-by":"crossref","unstructured":"Lugovic, S., Dundjer, I., Horvat, M.: Techniques and applications of emotion recognition in speech. In: 39th International Convention on Information and Communication Technology, Electronics and Microelectronics (mipro). IEEE 2016, 1278\u20131283 (2016)","DOI":"10.1109\/MIPRO.2016.7522336"},{"key":"3406_CR42","doi-asserted-by":"crossref","unstructured":"Cao, H., Cooper, D.G., Keutmann, M.K., Gur, R.C., Nenkova, A., Verma, R.: Crema-d: Crowd-sourced emotional multimodal actors dataset. IEEE Trans. Affect. Comput. 5(4), 377\u2013390 (2014)","DOI":"10.1109\/TAFFC.2014.2336244"},{"key":"3406_CR43","doi-asserted-by":"publisher","unstructured":"Pichora-Fuller, M.K., Dupuis, K.: Toronto emotional speech set (TESS), (2020). [Online]. Available: https:\/\/doi.org\/10.5683\/SP2\/E8H2MF","DOI":"10.5683\/SP2\/E8H2MF"},{"key":"3406_CR44","doi-asserted-by":"crossref","unstructured":"Wagner, J., Triantafyllopoulos, A., Wierstorf, H., Schmitt, M., Burkhardt, F., Eyben, F., Schuller, B.W.: Dawn of the transformer era in speech emotion recognition: Closing the valence gap. In: IEEE Transactions on Pattern Analysis and Machine Intelligence, pp. 1\u201313, (2023)","DOI":"10.1109\/TPAMI.2023.3263585"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03406-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-024-03406-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03406-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T10:32:23Z","timestamp":1723285943000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-024-03406-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,10]]},"references-count":44,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["3406"],"URL":"https:\/\/doi.org\/10.1007\/s11760-024-03406-8","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,10]]},"assertion":[{"value":"11 February 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 February 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 June 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 July 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}