{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,5,29]],"date-time":"2024-05-29T01:33:35Z","timestamp":1716946415245},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2014,10,4]],"date-time":"2014-10-04T00:00:00Z","timestamp":1412380800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2015,2]]},"DOI":"10.1007\/s00521-014-1708-8","type":"journal-article","created":{"date-parts":[[2014,10,3]],"date-time":"2014-10-03T07:31:30Z","timestamp":1412321490000},"page":"473-484","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Affect-insensitive speaker recognition systems via emotional speech clustering using prosodic features"],"prefix":"10.1007","volume":"26","author":[{"given":"Dongdong","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yubo","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaohui","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingchun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,10,4]]},"reference":[{"issue":"4","key":"1708_CR1","doi-asserted-by":"crossref","first-page":"277","DOI":"10.1016\/j.specom.2007.02.005","volume":"49","author":"AG Adami","year":"2007","unstructured":"Adami AG (2007) Modeling prosodic difference for speaker recognition. Speech Commun 49(4):277\u2013291","journal-title":"Speech Commun"},{"key":"1708_CR2","volume-title":"Towards an automatic classification of emotions in speech","author":"N Amir","year":"1998","unstructured":"Amir N, Ron S (1998) Towards an automatic classification of emotions in speech. ICSLP, Sydney"},{"key":"1708_CR3","first-page":"2821","volume-title":"Pitch-dependent GMM for Text-Independent Speaker Recognition Systems","author":"M Arcienega","year":"2001","unstructured":"Arcienega M, Drygajlo A (2001) Pitch-dependent GMM for Text-Independent Speaker Recognition Systems. EUROSPEECH, Scandinavia, pp 2821\u20132824"},{"key":"1708_CR4","doi-asserted-by":"crossref","unstructured":"Atal BS (1976) Automatic recognition of speakers from their voices. In: Proceedings of IEEE, pp 460\u2013475","DOI":"10.1109\/PROC.1976.10155"},{"issue":"1","key":"1708_CR5","doi-asserted-by":"crossref","first-page":"211","DOI":"10.1121\/1.381716","volume":"63","author":"JE Atkinson","year":"1978","unstructured":"Atkinson JE (1978) Correlation analysis of the physiological factors controlling fundamental voice frequency. J Acoust Soc Am 63(1):211\u2013222","journal-title":"J Acoust Soc Am"},{"key":"1708_CR6","doi-asserted-by":"crossref","DOI":"10.1109\/ICSLP.1996.608027","volume-title":"Automatic statistical analysis of the signal and prosodic signs of emotion in speech","author":"R Cowie","year":"1996","unstructured":"Cowie R, Douglas-Cowie EN (1996) Automatic statistical analysis of the signal and prosodic signs of emotion in speech. ICSLP, Philadelphia"},{"issue":"1","key":"1708_CR7","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1109\/79.911197","volume":"18","author":"R Cowie","year":"2001","unstructured":"Cowie R, Douglas-Cowie EN (2001) Emotion recognition in human\u2013computer interaction. IEEE Singal Process Mag 18(1):32\u201380","journal-title":"IEEE Singal Process Mag"},{"key":"1708_CR8","volume-title":"Towards real life application in emotion recognition","author":"K Daniel","year":"2004","unstructured":"Daniel K, Raquel T, Thomas K, Beate M (2004) Towards real life application in emotion recognition. ADS, Kloster Irsee"},{"key":"1708_CR9","volume-title":"Emotion-state conversion for speaker recognition","author":"L Dongdong","year":"2005","unstructured":"Dongdong L, Yingchun Y, Zhaohui W (2005) Emotion-state conversion for speaker recognition. ACII, Beijing"},{"key":"1708_CR10","unstructured":"Dongdong L, Yingchun Y (2009) Emotional speech clustering based robust speaker recognition system. In: 2nd international Congress on image and signal processing, pp 4576\u20134580"},{"issue":"2","key":"1708_CR11","doi-asserted-by":"crossref","first-page":"521","DOI":"10.1016\/0167-6393(91)90055-X","volume":"10","author":"G Fant","year":"1991","unstructured":"Fant G, Kruckenberg A, Nord L (1991) Prosodic and segmental speaker variations. Speech Commun 10(2):521\u2013531","journal-title":"Speech Commun"},{"issue":"2","key":"1708_CR12","first-page":"412","volume":"97","author":"RW Frick","year":"1985","unstructured":"Frick RW (1985) Communicating emotion: the role of prosodic features. Psychological 97(2):412\u2013429","journal-title":"Psychological"},{"issue":"4","key":"1708_CR13","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1109\/79.317924","volume":"11","author":"H Gish","year":"1994","unstructured":"Gish H, Schmidt N (1994) Text-independent speaker identification. IEEE Singal Process Mag 11(4):18\u201332","journal-title":"IEEE Singal Process Mag"},{"key":"1708_CR14","volume-title":"Towards combining pitch and MFCC for speaker identification systems","author":"E Hassan","year":"2001","unstructured":"Hassan E, Jean R (2001) Towards combining pitch and MFCC for speaker identification systems. EUROSPEECH, Aalborg"},{"key":"1708_CR15","unstructured":"Hirschberg J (1999) Communication and prosody: functional aspects of prosody. In: Proceedings of the ESCA workshop dialogue and prosody, pp 7\u201315"},{"key":"1708_CR16","volume-title":"Modeling dynamic prosodic variation for speaker verification","author":"S Kemal","year":"1998","unstructured":"Kemal S, Elizabeth S, Larry H, Mitchel W (1998) Modeling dynamic prosodic variation for speaker verifiction. ICSLP, Sydney"},{"key":"1708_CR17","unstructured":"Klasmeyer G, Johnstone T, Banziger T, Sappok C, Scherer KR (2000) Emotional voice variability in speaker verification. In: The ISCA workshop on speech and emotion, Newcastle, Northern Ireland, UK, pp 213\u2013218"},{"issue":"2","key":"1708_CR18","doi-asserted-by":"crossref","first-page":"820","DOI":"10.1121\/1.398894","volume":"87","author":"DH Klatt","year":"1990","unstructured":"Klatt DH, Klatt LC (1990) Analysis, synthesis, and perception of voice quality variations among female and male talkers. J Acoust Soc Am 87(2):820\u2013857","journal-title":"J Acoust Soc Am"},{"issue":"5","key":"1708_CR19","doi-asserted-by":"crossref","first-page":"58","DOI":"10.1109\/79.536825","volume":"13","author":"RJ Mammone","year":"1996","unstructured":"Mammone RJ, Zhang XY, Ramachandran RP (1996) Robust speaker recognition. IEEE Singal Process Mag 13(5):58\u201370","journal-title":"IEEE Singal Process Mag"},{"issue":"3","key":"1708_CR20","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1016\/S0167-6393(98)00010-7","volume":"24","author":"KP Markov","year":"1998","unstructured":"Markov KP, Nakagawa S (1998) Text-independent speaker recognition using non-linear frame likelihood transformation. Speech Commun 24(3):193\u2013209","journal-title":"Speech Commun"},{"key":"1708_CR21","volume-title":"The DET curve in assessment of detection task performance","author":"A Martin","year":"1997","unstructured":"Martin A, Doddington G, Kamm T, Ordowski M, Przybocki M (1997) The DET curve in assessment of detection task performance. EUROSPEECH, Rhodes"},{"issue":"1\u20132","key":"1708_CR22","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1016\/0167-6393(95)00011-C","volume":"17","author":"T Matsui","year":"1995","unstructured":"Matsui T, Furui S (1995) Likelihood normalization for speaker verification using a phoneme- and speaker-independent model. Speech Commun 17(1\u20132):109\u2013116","journal-title":"Speech Commun"},{"key":"1708_CR23","volume-title":"Modeling of variations in cepstral coefficients caused by Fo changes and its application to Speech Processing","author":"N Minematsu","year":"1998","unstructured":"Minematsu N, Nakagawa S (1998) Modeling of variations in cepstral coefficients caused by Fo changes and its application to Speech Processing. ICSLP, Sydney, Australia"},{"key":"1708_CR24","volume-title":"Emotional speech synthesis: from speech database to TTS","author":"JM Montero","year":"1998","unstructured":"Montero JM, Gutierrez-Arriola JM, Palazuelos S, Enriquez E, Aguilera S, Pardo JM (1998) Emotional speech synthesis: from speech database to TTS. ICSLP, Sydney"},{"key":"1708_CR25","volume-title":"Synthesizing emotions in speech: Is it time to get excited?","author":"IR Murray","year":"1996","unstructured":"Murray IR, Arnott JL (1996) Synthesizing emotions in speech: Is it time to get excited?. ICASSP, Philadelphia"},{"issue":"2","key":"1708_CR26","doi-asserted-by":"crossref","first-page":"107","DOI":"10.1016\/j.csl.2007.06.001","volume":"22","author":"IR Murray","year":"2008","unstructured":"Murray IR, Arnott JL (2008) Applying an analysis of acted vocal emotions to improve the simulation of synthetic speech. Comput Speech Lang 22(2):107\u2013129","journal-title":"Comput Speech Lang"},{"key":"1708_CR27","volume-title":"Using prosodic and conversational features for high-performance speaker recognition","author":"B Peskin","year":"2003","unstructured":"Peskin B, Navratil J, Abramson J, Jones D, Reynolds D, Xiang B (2003) Using prosodic and conversational features for high-performance speaker recognition. ICASSP, HongKong"},{"key":"1708_CR28","volume-title":"Exploiting glottal information in speaker recognition using parallel GMM","author":"Y Pu","year":"2005","unstructured":"Pu Y, Yingchun Y, Zhaohui W (2005) Exploiting glottal information in speaker recognition using parallel GMM. AVBPA, Hilton Rye Town"},{"key":"1708_CR29","unstructured":"Reynolds DA (1992) A Gaussian mixture modeling approach to text independent speaker identification. Georgia Institute of Technology"},{"key":"1708_CR30","first-page":"53","volume-title":"Channel robust speaker verification via feature mapping","author":"DA Reynolds","year":"2003","unstructured":"Reynolds DA (2003) Channel robust speaker verification via feature mapping. ICASSP, Hong Kong, pp 53\u201356"},{"key":"1708_CR31","volume-title":"The SuperSID Project: exploiting high-level information for high-accuracy speaker recognition","author":"DA Reynolds","year":"2003","unstructured":"Reynolds DA (2003) The SuperSID Project: exploiting high-level information for high-accuracy speaker recognition. ICASSP, HongKong"},{"key":"1708_CR32","volume-title":"A cross-cultural investigation of emotion inferences from voice and speech: implication for speech technology","author":"KR Scherer","year":"2000","unstructured":"Scherer KR (2000) A cross-cultural investigation of emotion inferences from voice and speech: implicationfor speech technology. ICSLP, Beijing"},{"issue":"1\u20132","key":"1708_CR33","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1016\/S0167-6393(02)00084-5","volume":"40","author":"KR Scherer","year":"2003","unstructured":"Scherer KR (2003) Vocal communication of emotion: a review of research paradigms. Speech Commun 40(1\u20132):227\u2013256","journal-title":"Speech Commun"},{"key":"1708_CR34","volume-title":"Can automatic speaker verification be improved by training the algorithms on emotional speech?","author":"KR Scherer","year":"2000","unstructured":"Scherer KR, Johnstone T, Klasmeyer G (2000) Can automatic speaker verification be improved by training the algorithms on emotional speech?. ICSLP, Beijing"},{"key":"1708_CR35","unstructured":"Scherer KR, Johnstone T, Banziger T (1998) Verification of emotionally stressed speakers: the problem of individual differences. SPECOM, pp 233\u2013238"},{"key":"1708_CR36","doi-asserted-by":"crossref","unstructured":"Schroder M (2001) Emotional speech synthesis: a review. EUROSPEECH, pp 561\u2013564","DOI":"10.21437\/Eurospeech.2001-150"},{"key":"1708_CR37","volume-title":"Integrated pitch and MFCC extraction for speech reconstruction and speech recognition applications","author":"X Shao","year":"2003","unstructured":"Shao X, Milner B, Cox S (2003) Integrated pitch and MFCC extraction for speech reconstruction and speech recognition applications. Eurospeech, Geneva"},{"issue":"6","key":"1708_CR38","doi-asserted-by":"crossref","first-page":"871","DOI":"10.1109\/29.1598","volume":"36","author":"FK Soong","year":"1988","unstructured":"Soong FK, Rosenberg AE (1988) On the use of instantaneous and transitional spectral information in speaker recognition. IEEE Trans Acoust Speech Signal Process 36(6):871\u2013879","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"key":"1708_CR39","volume-title":"Improving speaker recognition by training on emotion-added models","author":"W Tian","year":"2005","unstructured":"Tian W, Yingchun Y, Zhaohui W, Dongdong L (2005) Improving speaker recognition by training on emotion-added models. ACII, Beijing"},{"key":"1708_CR40","unstructured":"Tian W, Yingchun Y, Zhaohui W, Dongdong L (2006) MASC: a speech corpus in mandarin for emotion analysis and affective speaker recognition.Odyssey, San Juan, Puerto Rico, pp 1\u20135"},{"issue":"9","key":"1708_CR41","doi-asserted-by":"crossref","first-page":"1162","DOI":"10.1016\/j.specom.2006.04.003","volume":"48","author":"D Ververidis","year":"2004","unstructured":"Ververidis D, Kotropoulos C (2004) Emotional speech recognition: resources, features, and methods. Speech Commun 48(9):1162\u20131181","journal-title":"Speech Commun"},{"key":"1708_CR42","unstructured":"Wei W, Thomas FZ, Xu MX, HuanJun B (2006) Study on speaker verification on emotional speech. Interspeech, pp 2102\u20132105"},{"key":"1708_CR43","doi-asserted-by":"crossref","DOI":"10.1109\/ICASSP.2006.1660107","volume-title":"Rules based feature modification for affective speaker recognition","author":"W Zhaohui","year":"2006","unstructured":"Zhaohui W, Dongdong L, Yingchun Y (2006) Rules based feature modification for affective speaker recognition. ICASSP, Toulouse"},{"key":"1708_CR44","doi-asserted-by":"crossref","unstructured":"Zilca RD, Navratil J, Ramaswamy GN (2003) SynPitch: a pseudo pitch synchronous algorithm for speaker recognition, Eurospeech, pp 2649\u20132652","DOI":"10.21437\/Eurospeech.2003-723"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-014-1708-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-014-1708-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-014-1708-8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,17]],"date-time":"2023-07-17T00:26:26Z","timestamp":1689553586000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-014-1708-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,10,4]]},"references-count":44,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2015,2]]}},"alternative-id":["1708"],"URL":"https:\/\/doi.org\/10.1007\/s00521-014-1708-8","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,10,4]]}}}