{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T15:23:41Z","timestamp":1785857021931,"version":"3.56.0"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,3,28]],"date-time":"2022-03-28T00:00:00Z","timestamp":1648425600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,3,28]],"date-time":"2022-03-28T00:00:00Z","timestamp":1648425600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100006261","name":"Taif University","doi-asserted-by":"publisher","award":["TURSP-2020\/36"],"award-info":[{"award-number":["TURSP-2020\/36"]}],"id":[{"id":"10.13039\/501100006261","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2022,5]]},"DOI":"10.1007\/s00138-022-01294-x","type":"journal-article","created":{"date-parts":[[2022,3,28]],"date-time":"2022-03-28T15:03:58Z","timestamp":1648479838000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":49,"title":["Convolutional neural network-based cross-corpus speech emotion recognition with data augmentation and features fusion"],"prefix":"10.1007","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1129-6006","authenticated-orcid":false,"given":"Rashid","family":"Jahangir","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying Wah","family":"Teh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ghulam","family":"Mujtaba","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Roobaea","family":"Alroobaea","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zahid Hussain","family":"Shaikh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ihsan","family":"Ali","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,3,28]]},"reference":[{"key":"1294_CR1","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1016\/j.ins.2019.09.005","volume":"509","author":"L Chen","year":"2020","unstructured":"Chen, L., Su, W., Feng, Y., Wu, M., She, J., et al.: Two-layer fuzzy multiple random forest for speech emotion recognition in human-robot interaction. Inf. Sci. 509, 150\u2013163 (2020)","journal-title":"Inf. Sci."},{"issue":"1","key":"1294_CR2","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1016\/j.vrih.2020.11.006","volume":"3","author":"W Zheng","year":"2021","unstructured":"Zheng, W., Zheng, W., Zong, Y.: Multi-scale discrepancy adversarial network for crosscorpus speech emotion recognition. Virtual Real. Intell. Hardw. 3(1), 65\u201375 (2021)","journal-title":"Virtual Real. Intell. Hardw."},{"issue":"4","key":"1294_CR3","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1016\/0167-6393(95)00007-B","volume":"16","author":"JH Hansen","year":"1995","unstructured":"Hansen, J.H., Cairns, D.A.: Icarus: Source generator based real-time recognition of speech in noisy stressful and lombard effect environments\u2606. Speech Commun. 16(4), 391\u2013422 (1995)","journal-title":"Speech Commun."},{"issue":"1","key":"1294_CR4","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1007\/s10772-020-09672-4","volume":"23","author":"A Koduru","year":"2020","unstructured":"Koduru, A., Valiveti, H.B., Budati, A.K.: Feature extraction algorithms to improve the speech emotion recognition rate. Int. J. Speech Technol. 23(1), 45\u201355 (2020)","journal-title":"Int. J. Speech Technol."},{"key":"1294_CR5","doi-asserted-by":"crossref","unstructured":"Schuller, B., Rigoll, G., Lang, M.: Speech emotion recognition combining acoustic features and linguistic information in a hybrid support vector machine-belief network architecture. In: 2004 IEEE International Conference on Acoustics, Speech, and Signal Processing, pp. I-577 (2004)","DOI":"10.1109\/ICASSP.2004.1326051"},{"key":"1294_CR6","doi-asserted-by":"crossref","unstructured":"Spencer, C., Ko\u00e7, \u0130.A., Suga, C., Lee, A., Dhareshwar, A.M., et al.: A comparison of unimodal and multimodal measurements of driver stress in real-world driving conditions. (2020)","DOI":"10.31234\/osf.io\/en5r3"},{"issue":"7","key":"1294_CR7","doi-asserted-by":"publisher","first-page":"829","DOI":"10.1109\/10.846676","volume":"47","author":"DJ France","year":"2000","unstructured":"France, D.J., Shiavi, R.G., Silverman, S., Silverman, M., Wilkes, M.: Acoustical properties of speech as indicators of depression and suicidal risk. IEEE Trans. Biomed. Eng. 47(7), 829\u2013837 (2000)","journal-title":"IEEE Trans. Biomed. Eng."},{"key":"1294_CR8","doi-asserted-by":"publisher","first-page":"103775","DOI":"10.1016\/j.engappai.2020.103775","volume":"94","author":"MZ Uddin","year":"2020","unstructured":"Uddin, M.Z., Nilsson, E.G.: Emotion recognition using speech and neural structured learning to facilitate edge intelligence. Eng. Appl. Artif. Intell. 94, 103775 (2020)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"1294_CR9","first-page":"1","volume":"80","author":"R Jahangir","year":"2021","unstructured":"Jahangir, R., Teh, Y.W., Hanif, F., Mujtaba, G.: Deep learning approaches for speech emotion recognition: state of the art and research challenges. Multimed. Tools Appl. 80, 1\u201366 (2021)","journal-title":"Multimed. Tools Appl."},{"issue":"10","key":"1294_CR10","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid, O., Mohamed, A.-R., Jiang, H., Deng, L., Penn, G., et al.: Convolutional neural networks for speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(10), 1533\u20131545 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"1294_CR11","doi-asserted-by":"crossref","unstructured":"Trigeorgis, G., Ringeval, F., Brueckner, R., Marchi, E., Nicolaou, M.A., et al.: Adieu features? end-to-end speech emotion recognition using a deep convolutional recurrent network. In: 2016 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 5200\u20135204 (2016).","DOI":"10.1109\/ICASSP.2016.7472669"},{"key":"1294_CR12","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural. Inf. Process. Syst. 25, 1097\u20131105 (2012)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1294_CR13","doi-asserted-by":"crossref","unstructured":"Fu, L., Mao, X., Chen, L.: Speaker independent emotion recognition based on SVM\/HMMs fusion system. In: 2008 international conference on audio, language and image processing, pp. 61\u201365 (2008).","DOI":"10.1109\/ICINIS.2008.64"},{"key":"1294_CR14","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1016\/j.specom.2019.12.001","volume":"116","author":"MB Ak\u00e7ay","year":"2020","unstructured":"Ak\u00e7ay, M.B., O\u011fuz, K.: Speech emotion recognition: Emotional models, databases, features, preprocessing methods, supporting modalities, and classifiers. Speech Commun. 116, 56\u201376 (2020)","journal-title":"Speech Commun."},{"key":"1294_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11042-020-10329-2","volume":"80","author":"MD Pawar","year":"2021","unstructured":"Pawar, M.D., Kokate, R.D.: Convolution neural network based automatic speech emotion recognition using Mel-frequency Cepstrum coefficients. Multim. Tools Appl. 80, 1\u201325 (2021)","journal-title":"Multim. Tools Appl."},{"key":"1294_CR16","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/j.specom.2020.12.009","volume":"127","author":"S Zhang","year":"2021","unstructured":"Zhang, S., Tao, X., Chuang, Y., Zhao, X.: Learning deep multimodal affective features for spontaneous speech emotion recognition. Speech Commun. 127, 73\u201381 (2021)","journal-title":"Speech Commun."},{"key":"1294_CR17","doi-asserted-by":"publisher","first-page":"101894","DOI":"10.1016\/j.bspc.2020.101894","volume":"59","author":"D Issa","year":"2020","unstructured":"Issa, D., Demirci, M.F., Yazici, A.: Speech emotion recognition with deep convolutional neural networks. Biomed. Signal Process. Control 59, 101894 (2020)","journal-title":"Biomed. Signal Process. Control"},{"key":"1294_CR18","doi-asserted-by":"publisher","first-page":"79861","DOI":"10.1109\/ACCESS.2020.2990405","volume":"8","author":"M Sajjad","year":"2020","unstructured":"Sajjad, M., Kwon, S.: Clustering-based speech emotion recognition by incorporating learned features and deep BiLSTM. IEEE Access 8, 79861\u201379875 (2020)","journal-title":"IEEE Access"},{"issue":"5","key":"1294_CR19","doi-asserted-by":"publisher","first-page":"5571","DOI":"10.1007\/s11042-017-5292-7","volume":"78","author":"AM Badshah","year":"2019","unstructured":"Badshah, A.M., Rahim, N., Ullah, N., Ahmad, J., Muhammad, K., et al.: Deep features-based speech emotion recognition for smart affective services. Multimed. Tools Appl. 78(5), 5571\u20135589 (2019)","journal-title":"Multimed. Tools Appl."},{"key":"1294_CR20","doi-asserted-by":"publisher","first-page":"221640","DOI":"10.1109\/ACCESS.2020.3043201","volume":"8","author":"MB Er","year":"2020","unstructured":"Er, M.B.: A novel approach for classification of speech emotions based on deep and acoustic features. IEEE Access 8, 221640\u2013221653 (2020)","journal-title":"IEEE Access"},{"issue":"4","key":"1294_CR21","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1016\/S0167-6393(03)00099-2","volume":"41","author":"TL Nwe","year":"2003","unstructured":"Nwe, T.L., Foo, S.W., De Silva, L.C.: Speech emotion recognition using hidden Markov models. Speech Commun. 41(4), 603\u2013623 (2003)","journal-title":"Speech Commun."},{"issue":"4","key":"1294_CR22","doi-asserted-by":"publisher","first-page":"290","DOI":"10.1007\/s005210070006","volume":"9","author":"J Nicholson","year":"2000","unstructured":"Nicholson, J., Takahashi, K., Nakatsu, R.: Emotion recognition in speech using neural networks. Neural Comput. Appl. 9(4), 290\u2013296 (2000)","journal-title":"Neural Comput. Appl."},{"issue":"2","key":"1294_CR23","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/s10772-017-9396-2","volume":"20","author":"F Noroozi","year":"2017","unstructured":"Noroozi, F., Sapi\u0144ski, T., Kami\u0144ska, D., Anbarjafari, G.: Vocal-based emotion recognition using random forests and decision tree. Int. J. Speech Technol. 20(2), 239\u2013246 (2017)","journal-title":"Int. J. Speech Technol."},{"key":"1294_CR24","doi-asserted-by":"publisher","first-page":"32187","DOI":"10.1109\/ACCESS.2020.2973541","volume":"8","author":"R Jahangir","year":"2020","unstructured":"Jahangir, R., Teh, Y.W., Memon, N.A., Mujtaba, G., Zareei, M., et al.: Text-independent speaker identification through feature fusion and deep neural network. IEEE Access 8, 32187\u201332202 (2020)","journal-title":"IEEE Access"},{"key":"1294_CR25","doi-asserted-by":"publisher","first-page":"127081","DOI":"10.1109\/ACCESS.2021.3110992","volume":"9","author":"RH Aljuhani","year":"2021","unstructured":"Aljuhani, R.H., Alshutayri, A., Alahdal, S.: Arabic speech emotion recognition from saudi dialect corpus. IEEE Access 9, 127081\u2013127085 (2021)","journal-title":"IEEE Access"},{"key":"1294_CR26","doi-asserted-by":"crossref","unstructured":"Burkhardt, F., Paeschke, A., Rolfes, M., Sendlmeier, W.F., Weiss, B.: A database of German emotional speech. In: Ninth European Conference on Speech Communication and Technology (2005).","DOI":"10.21437\/Interspeech.2005-446"},{"issue":"5","key":"1294_CR27","doi-asserted-by":"publisher","first-page":"e0196391","DOI":"10.1371\/journal.pone.0196391","volume":"13","author":"SR Livingstone","year":"2018","unstructured":"Livingstone, S.R., Russo, F.A.: The Ryerson audio-visual database of emotional speech and song (RAVDESS): a dynamic, multimodal set of facial and vocal expressions in north American english. PLoS ONE 13(5), e0196391 (2018)","journal-title":"PLoS ONE"},{"key":"1294_CR28","volume-title":"Surrey audio-visual expressed emotion (savee) database","author":"P Jackson","year":"2014","unstructured":"Jackson, P., Haq, S.: Surrey audio-visual expressed emotion (savee) database. University of Surrey, Guildford, UK (2014)"},{"key":"1294_CR29","unstructured":"DeVries, T., Taylor, G.W.: Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552 (2017)."},{"issue":"245","key":"1294_CR30","first-page":"1","volume":"21","author":"S Chen","year":"2020","unstructured":"Chen, S., Dobriban, E., Lee, J.H.: A group-theoretic framework for data augmentation. J. Mach. Learn. Res. 21(245), 1\u201371 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"1294_CR31","unstructured":"Hannun, A., Case, C., Casper, J., Catanzaro, B., Diamos, G., et al.: Deep speech: scaling up end-to-end speech recognition. arXiv preprint arXiv:1412.5567 (2014)."},{"key":"1294_CR32","doi-asserted-by":"crossref","unstructured":"Wei, S., Zou, S., Liao, F.: A comparison on data augmentation methods based on deep learning for audio classification. In: Journal of Physics: Conference Series, p. 012085, (2020).","DOI":"10.1088\/1742-6596\/1453\/1\/012085"},{"issue":"10","key":"1294_CR33","doi-asserted-by":"publisher","first-page":"78","DOI":"10.1145\/2347736.2347755","volume":"55","author":"P Domingos","year":"2012","unstructured":"Domingos, P.: A few useful things to know about machine learning. Commun. ACM 55(10), 78\u201387 (2012)","journal-title":"Commun. ACM"},{"key":"1294_CR34","doi-asserted-by":"crossref","unstructured":"McFee, B., Raffel, C., Liang, D., Ellis, D.P., McVicar, M., et al.: librosa: audio and music signal analysis in python. In: Proceedings of the 14th Python in Science Conference, pp. 18\u201325 (2015).","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"1294_CR35","doi-asserted-by":"crossref","unstructured":"Palo, H.K., Chandra, M., Mohanty, M.N.: Recognition of human speech emotion using variants of mel-frequency cepstral coefficients. In: Advances in Systems, Control and Automation. Springer, pp. 491-498 (2018)","DOI":"10.1007\/978-981-10-4762-6_47"},{"issue":"41","key":"1294_CR36","doi-asserted-by":"publisher","first-page":"31265","DOI":"10.1007\/s11042-020-09580-4","volume":"79","author":"SR Shahamiri","year":"2020","unstructured":"Shahamiri, S.R., Thabtah, F.: An investigation towards speaker identification using a single-sound-frame. Multimed. Tools Appl. 79(41), 31265\u201331281 (2020)","journal-title":"Multimed. Tools Appl."},{"issue":"10","key":"1294_CR37","doi-asserted-by":"publisher","first-page":"15511","DOI":"10.1007\/s11042-020-10381-y","volume":"80","author":"H-C Wang","year":"2021","unstructured":"Wang, H.-C., Syu, S.-W., Wongchaisuwat, P.: A method of music autotagging based on audioand lyrics. Multimed. Tools Appl. 80(10), 15511\u201315539 (2021)","journal-title":"Multimed. Tools Appl."},{"key":"1294_CR38","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1007\/978-0-387-77592-0_17","volume-title":"Fundamentals of Speaker Recognition","author":"H Beigi","year":"2011","unstructured":"Beigi, H.: Speaker recognition. In: Fundamentals of Speaker Recognition, pp. 543\u2013559. Springer, Boston, MA (2011). https:\/\/doi.org\/10.1007\/978-0-387-77592-0_17"},{"key":"1294_CR39","doi-asserted-by":"publisher","unstructured":"Harte, C., Sandler, M., Gasser, M.: Detecting harmonic change in musical audio. Presented at the Proceedings of the 1st ACM workshop on Audio and music computing multimedia, Santa Barbara, California, USA, 2006. [Online]. https:\/\/doi.org\/10.1145\/1178723.1178727.","DOI":"10.1145\/1178723.1178727"},{"key":"1294_CR40","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/j.eswa.2018.03.056","volume":"105","author":"HF Nweke","year":"2018","unstructured":"Nweke, H.F., Teh, Y.W., Al-Garadi, M.A., Alo, U.R.: Deep learning algorithms for human activity recognition using mobile and wearable sensor networks: state of the art and research challenges. Expert Syst. Appl. 105, 233\u2013261 (2018)","journal-title":"Expert Syst. Appl."},{"key":"1294_CR41","doi-asserted-by":"publisher","first-page":"365","DOI":"10.1007\/s11257-019-09248-1","volume":"30","author":"E Garcia-Ceja","year":"2020","unstructured":"Garcia-Ceja, E., Riegler, M., Kvernberg, A.K., Torresen, J.: User-adaptive models for activity andemotion recognition using deep transfer learning and data augmentation. User Model User-Adap Inter. 30, 365\u2013393 (2020)","journal-title":"User Model User-AdapInter."},{"issue":"3793","key":"1294_CR42","first-page":"3804","volume":"23","author":"W Nie","year":"2020","unstructured":"Nie, W., Ren, M., Nie, J., Zhao, S.: C-GCN: correlation based graph convolutional network for audio-video emotion recognition. IEEE Trans. Multimed. 23(3793), 3804 (2020)","journal-title":"IEEE Trans. Multimed."},{"key":"1294_CR43","unstructured":"Gholamy, A., Kreinovich, V., Kosheleva, O.: Why 70\/30 or 80\/20 relation between training and testing sets: a pedagogical explanation. Departmental Technical Reports (CS) 1209 (2018). https:\/\/scholarworks.utep.edu\/cgi\/viewcontent.cgi?article=2202&context=cs_techrep"},{"issue":"5","key":"1294_CR44","doi-asserted-by":"publisher","first-page":"479","DOI":"10.3390\/e21050479","volume":"21","author":"N Hajarolasvadi","year":"2019","unstructured":"Hajarolasvadi, N., Demirel, H.: 3D CNN-based speech emotion recognition using K-means clustering and spectrograms. Entropy 21(5), 479 (2019)","journal-title":"Entropy"},{"issue":"21","key":"1294_CR45","doi-asserted-by":"publisher","first-page":"6008","DOI":"10.3390\/s20216008","volume":"20","author":"M Farooq","year":"2020","unstructured":"Farooq, M., Hussain, F., Baloch, N.K., Raja, F.R., Yu, H., et al.: Impact of feature selection algorithm on speech emotion recognition using deep convolutional neural network. Sensors 20(21), 6008 (2020)","journal-title":"Sensors"},{"issue":"8","key":"1294_CR46","doi-asserted-by":"publisher","first-page":"e0220386","DOI":"10.1371\/journal.pone.0220386","volume":"14","author":"P Heracleous","year":"2019","unstructured":"Heracleous, P., Yoneyama, A.: A comprehensive study on bilingual and multilingual speech emotion recognition using a two-pass classification scheme. PLoS ONE 14(8), e0220386 (2019)","journal-title":"PLoS ONE"},{"key":"1294_CR47","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1016\/j.neunet.2021.03.013","volume":"141","author":"Z Zhao","year":"2021","unstructured":"Zhao, Z., Li, Q., Zhang, Z., Cummins, N., Wang, H., et al.: Combining a parallel 2D CNN with a self-attention Dilated Residual Network for CTC-Based discrete speech emotion recognition. Neural Netw. 141, 52\u201360 (2021)","journal-title":"Neural Netw."},{"key":"1294_CR48","doi-asserted-by":"publisher","first-page":"107101","DOI":"10.1016\/j.asoc.2021.107101","volume":"102","author":"S Kwon","year":"2021","unstructured":"Kwon, S.: Att-Net: Enhanced emotion recognition system using lightweight self-attention module. Appl. Soft Comput. 102, 107101 (2021)","journal-title":"Appl. Soft Comput."}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-022-01294-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-022-01294-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-022-01294-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,21]],"date-time":"2024-09-21T03:55:02Z","timestamp":1726890902000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-022-01294-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,28]]},"references-count":48,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,5]]}},"alternative-id":["1294"],"URL":"https:\/\/doi.org\/10.1007\/s00138-022-01294-x","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,3,28]]},"assertion":[{"value":"5 April 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 December 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 February 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"41"}}