{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T18:51:05Z","timestamp":1774551065776,"version":"3.50.1"},"reference-count":26,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2017,8,1]],"date-time":"2017-08-01T00:00:00Z","timestamp":1501545600000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"published-print":{"date-parts":[[2017,12]]},"DOI":"10.1186\/s13636-017-0116-2","type":"journal-article","created":{"date-parts":[[2017,8,1]],"date-time":"2017-08-01T15:05:54Z","timestamp":1501599954000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["Emotional voice conversion using neural networks with arbitrary scales F0 based on wavelet transform"],"prefix":"10.1186","volume":"2017","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4173-6319","authenticated-orcid":false,"given":"Zhaojie","family":"Luo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinhui","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tetsuya","family":"Takiguchi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yasuo","family":"Ariki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,8,1]]},"reference":[{"key":"116_CR1","unstructured":"S Mori, T Moriyama, S Ozawa, in Proc. ICME. Emotional speech synthesis using subspace constraints in prosody, (2006), pp. 1093\u20131096."},{"key":"116_CR2","unstructured":"R Aihara, T Takiguchi, Y Ariki, in Proc. SLPAT. Individuality-preserving voice conversion for articulation disorders using dictionary selective non-negative matrix factorization, (2014), pp. 29\u201337."},{"issue":"1","key":"116_CR3","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1515\/lp-2013-0003","volume":"4","author":"J Krivokapi\u0107","year":"2013","unstructured":"J Krivokapi\u0107, Rhythm and convergence between speakers of american and indian english. Lab. Phonol. 4(1), 39\u201365 (2013).","journal-title":"Lab. Phonol"},{"key":"116_CR4","doi-asserted-by":"crossref","unstructured":"T Raitio, L Juvela, A Suni, M Vainio, P Alku, in Proc. Sixteenth Annual Conference of the International Speech Communication Association. Phase perception of the glottal excitation of vocoded speech, (2015).","DOI":"10.21437\/Interspeech.2015-112"},{"key":"116_CR5","doi-asserted-by":"crossref","unstructured":"Z-W Shuang, R Bakis, S Shechtman, D Chazan, Y Qin, in Proc. Ninth International Conference on Spoken Language Processing. Frequency warping based on mapping formant parameters, (2006).","DOI":"10.21437\/Interspeech.2006-588"},{"key":"116_CR6","unstructured":"D Erro, A Moreno, in Proc. Interspeech. Weighted frequency warping for voice conversion, (2007), pp. 1965\u20131968."},{"issue":"8","key":"116_CR7","doi-asserted-by":"crossref","first-page":"2222","DOI":"10.1109\/TASL.2007.907344","volume":"15","author":"T Toda","year":"2007","unstructured":"T Toda, AW Black, K Tokuda, Voice conversion based on maximum-likelihood estimation of spectral parameter trajectory. IEEE Trans. Audio Speech Lang. Process. 15(8), 2222\u20132235 (2007).","journal-title":"IEEE Trans. Audio Speech Lang. Process"},{"issue":"5","key":"116_CR8","doi-asserted-by":"crossref","first-page":"912","DOI":"10.1109\/TASL.2010.2041699","volume":"18","author":"E Helander","year":"2010","unstructured":"E Helander, T Virtanen, J Nurminen, M Gabbouj, Voice conversion using partial least squares regression. IEEE Trans. Audio Speech Lang. Process. 18(5), 912\u2013921 (2010).","journal-title":"IEEE Trans. Audio Speech Lang. Process"},{"key":"116_CR9","unstructured":"R Takashima, T Takiguchi, Y Ariki, in Proc. Spoken Language Technology Workshop (SLT). Exemplar-based voice conversion in noisy environment, (2012), pp. 313\u2013317."},{"key":"116_CR10","unstructured":"T Fukada, K Tokuda, T Kobayashi, S Imai, in Proc. ICASSP. An adaptive algorithm for mel-cepstral analysis of speech, (1992), pp. 137\u2013140."},{"key":"116_CR11","unstructured":"S Desai, EV Raghavendra, B Yegnanarayana, AW Black, K Prahallad, in Proc. ICASSP. Voice conversion using artificial neural networks, (2009), pp. 3893\u20133896."},{"key":"116_CR12","unstructured":"T Nakashika, R Takashima, T Takiguchi, Y Ariki, in Proc. INTERSPEECH. Voice conversion in high-order eigen space using deep belief nets, (2013), pp. 369\u2013372."},{"issue":"6","key":"116_CR13","doi-asserted-by":"crossref","first-page":"349","DOI":"10.1250\/ast.27.349","volume":"27","author":"H Kawahara","year":"2006","unstructured":"H Kawahara, Straight, exploitation of the other aspect of vocoder: perceptually isomorphic decomposition of speech sounds. Acoust. Sci. Technol. 27(6), 349\u2013353 (2006).","journal-title":"Acoust. Sci. Technol"},{"key":"116_CR14","doi-asserted-by":"crossref","first-page":"410","DOI":"10.1109\/FSKD.2007.347","volume":"4","author":"K Liu","year":"2007","unstructured":"K Liu, J Zhang, Y Yan, High quality voice conversion through phoneme-based linear mapping functions with straight for mandarin. Fuzzy Syst. Knowl. Discov. 4:, 410\u2013414 (2007). IEEE.","journal-title":"Fuzzy Syst. Knowl. Discov"},{"key":"116_CR15","doi-asserted-by":"crossref","unstructured":"MS Ribeiro, RA Clark, in ICASSP. A multi-level representation of f0 using the continuous wavelet transform and the discrete cosine transform (IEEE, 2015), pp. 4909\u20134913.","DOI":"10.1109\/ICASSP.2015.7178904"},{"key":"116_CR16","unstructured":"Z Luo, T Takiguchi, Y Ariki, in Proc. IEEE\/ACIS 15th International Conference on Computer and Information Science (ICIS). Emotional voice conversion using deep neural networks with mcc and f0 features, (2016), pp. 1\u20135."},{"key":"116_CR17","unstructured":"M Vainio, A Suni, D Aalto, et al, in Proc. TRASP 2013-Tools and Resources for the Analysys of Speech Prosody. Continuous wavelet transform for analysis of speech prosody, (2013)."},{"key":"116_CR18","unstructured":"AS Suni, D Aalto, T Raitio, P Alku, M Vainio, et al, in Proc. 8th ISCA Workshop on Speech Synthesis, Proceedings, Barcelona, August 31-September 2, 2013. Wavelets for intonation modeling in hmm speech synthesis, (2013)."},{"key":"116_CR19","doi-asserted-by":"crossref","unstructured":"H Ming, D Huang, M Dong, H Li, L Xie, S Zhang, in Affective Computing and Intelligent Interaction (ACII). Fundamental frequency modeling using wavelets for emotional voice conversion (IEEE, 2015), pp. 804\u2013809.","DOI":"10.1109\/ACII.2015.7344665"},{"key":"116_CR20","unstructured":"H Ming, D Huang, L Xie, J Wu, M Dong, H Li, in Proc. INTERSPEECH. Deep bidirectional LSTM modeling of timbre and prosody for emotional voice conversion, (2016), pp. 2453\u20132457."},{"key":"116_CR21","doi-asserted-by":"crossref","unstructured":"Z Luo, J Chen, T Nakashika, T Takiguchi, Y Ariki, in Proc. 9th ISCA Speech Synthesis Workshop. Emotional voice conversion using neural networks with different temporal scales of f0 based on wavelet transform, (2016).","DOI":"10.21437\/SSW.2016-23"},{"key":"116_CR22","unstructured":"T Toda, et al, Interlanguage phonology: acquisition of timing control and perceptual categorization of durational contrast in japanese (2013)."},{"key":"116_CR23","unstructured":"S Mallat, A wavelet tour of signal processing: the sparse way. Investigaci\u00f3n Operacional (Academic press, Elsevier, 2008)."},{"key":"116_CR24","doi-asserted-by":"crossref","unstructured":"H Kawanami, Y Iwami, T Toda, H Saruwatari, K Shikano, GMM-based voice conversion applied to emotional speech synthesis. IEEE Trans. Speech Audio Proc, 2401\u20132404 (2003).","DOI":"10.21437\/Eurospeech.2003-661"},{"issue":"9","key":"116_CR25","doi-asserted-by":"crossref","first-page":"1162","DOI":"10.1016\/j.specom.2006.04.003","volume":"48","author":"D Ververidis","year":"2006","unstructured":"D Ververidis, C Kotropoulos, Emotional speech recognition: resources, features, and methods. Speech Comm. 48(9), 1162\u20131181 (2006).","journal-title":"Speech Comm"},{"issue":"1","key":"116_CR26","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1186\/s13640-016-0140-7","volume":"2016","author":"J Chen","year":"2016","unstructured":"J Chen, Z Luo, T Takiguchi, Y Ariki, Multithreading cascade of surf for facial expression recognition. EURASIP J. Image Video Process. 2016(1), 37 (2016).","journal-title":"EURASIP J. Image Video Process"}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-017-0116-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1186\/s13636-017-0116-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-017-0116-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,24]],"date-time":"2023-08-24T20:08:57Z","timestamp":1692907737000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-017-0116-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,8,1]]},"references-count":26,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2017,12]]}},"alternative-id":["116"],"URL":"https:\/\/doi.org\/10.1186\/s13636-017-0116-2","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,8,1]]},"article-number":"18"}}