{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T05:12:53Z","timestamp":1744261973100,"version":"3.37.3"},"reference-count":67,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T00:00:00Z","timestamp":1570492800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T00:00:00Z","timestamp":1570492800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1007\/s10772-019-09643-4","type":"journal-article","created":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T18:12:10Z","timestamp":1570558330000},"page":"1007-1019","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A novel voice conversion approach using cascaded powerful cepstrum predictors with excitation and phase extracted from the target training space encoded as a KD-tree"],"prefix":"10.1007","volume":"22","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6345-5366","authenticated-orcid":false,"given":"Imen","family":"Ben Othmane","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joseph","family":"Di Martino","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ka\u00efs","family":"Ouni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,10,8]]},"reference":[{"issue":"2","key":"9643_CR1","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1250\/ast.11.71","volume":"11","author":"M Abe","year":"1990","unstructured":"Abe, M., Nakamura, S., Shikano, K., & Kuwabara, H. (1990). Voice conversion through vector quantization. Journal of the Acoustical Society of Japan (E), 11(2), 71\u201376.","journal-title":"Journal of the Acoustical Society of Japan (E)"},{"issue":"3","key":"9643_CR2","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1016\/S0167-6393(99)00015-1","volume":"28","author":"LM Arslan","year":"1999","unstructured":"Arslan, L. M. (1999). Speaker transformation algorithm using segmental codebooks (stasc) 1. Speech Communication, 28(3), 211\u2013226.","journal-title":"Speech Communication"},{"key":"9643_CR3","unstructured":"Arya, S. (1996). Nearest neighbor searching and applications. PhD thesis, University of Maryland, College Park."},{"key":"9643_CR4","unstructured":"Azarov, E., Petrovsky, A., & Zubrycki, P. (2010). Multi voice text to speech synthesis based on the instantaneous parametric voice conversion. In Signal processing algorithms, architectures, arrangements, and applications SPA 2010 (pp. 78\u201382). IEEE."},{"key":"9643_CR5","unstructured":"Beauregard, G.\u00a0T., Zhu, X., & Wyse, L. (2005). An efficient algorithm for real-time spectrogram inversion. In Proceedings of the 8th international conference on digital audio effects (pp. 116\u2013118)."},{"key":"9643_CR6","first-page":"1","volume":"22","author":"I Ben Othmane","year":"2018","unstructured":"Ben Othmane, I., Di Martino, J., & Ouni, K. (2018a). Enhancement of esophageal speech obtained by a voice conversion technique using time dilated fourier cepstra. International Journal of Speech Technology, 22, 1\u201312.","journal-title":"International Journal of Speech Technology"},{"key":"9643_CR7","doi-asserted-by":"crossref","unstructured":"Ben\u00a0Othmane, I., Di\u00a0Martino, J., & Ouni, K. (2018b). Improving the computational performance of standard gmm-based voice conversion systems used in real-time applications. In 2018 International conference on electronics, control, optimization and computer science (ICECOCS) (pp. 1\u20135). IEEE.","DOI":"10.1109\/ICECOCS.2018.8610514"},{"key":"9643_CR8","doi-asserted-by":"crossref","unstructured":"Charlier, M., Ohtani, Y., Toda, T., Moinet, A., & Dutoit, T. (2009). Cross-language voice conversion based on eigenvoices. In 10th Annual conference of the international speech communication association","DOI":"10.21437\/Interspeech.2009-488"},{"issue":"12","key":"9643_CR9","doi-asserted-by":"publisher","first-page":"1859","DOI":"10.1109\/TASLP.2014.2353991","volume":"22","author":"L-H Chen","year":"2014","unstructured":"Chen, L.-H., Ling, Z.-H., Liu, L.-J., & Dai, L.-R. (2014). Voice conversion using deep neural networks with layer-wise generative training. IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP), 22(12), 1859\u20131872.","journal-title":"IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP)"},{"key":"9643_CR10","first-page":"3052","volume":"87","author":"L-H Chen","year":"2013","unstructured":"Chen, L.-H., Ling, Z.-H., Song, Y., & Dai, L.-R. (2013). Joint spectral distribution modeling using restricted Boltzmann machines for voice conversion. Interspeech, 87, 3052\u20133056.","journal-title":"Interspeech"},{"key":"9643_CR11","doi-asserted-by":"crossref","unstructured":"Chen, L.-H., Yang, C.-Y., Ling, Z.-H., Jiang, Y., Dai, L.-R., Hu, Y., & Wang, R.-H. (2011). The USTC system for blizzard challenge 2011. In Blizzard challenge workshop.","DOI":"10.21437\/Blizzard.2011-10"},{"issue":"5","key":"9643_CR12","doi-asserted-by":"publisher","first-page":"954","DOI":"10.1109\/TASL.2010.2047683","volume":"18","author":"S Desai","year":"2010","unstructured":"Desai, S., Black, A. W., Yegnanarayana, B., & Prahallad, K. (2010). Spectral mapping using artificial neural networks for voice conversion. IEEE Transactions on Audio, Speech, and Language Processing, 18(5), 954\u2013964.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9643_CR13","doi-asserted-by":"crossref","unstructured":"Desai, S., Raghavendra, E.\u00a0V., Yegnanarayana, B., Black, A.\u00a0W., & Prahallad, K. (2009). Voice conversion using artificial neural networks. In IEEE International Conference on Acoustics, Speech and Signal Processing, 2009. ICASSP 2009 (pp. 3893\u20133896). IEEE.","DOI":"10.1109\/ICASSP.2009.4960478"},{"key":"9643_CR14","doi-asserted-by":"crossref","unstructured":"Deza, M.\u00a0M., & Deza, E. (2009). Encyclopedia of distances. In Encyclopedia of distances (pp. 1\u2013583). Springer, Berlin","DOI":"10.1007\/978-3-642-00234-2_1"},{"issue":"1","key":"9643_CR15","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TASLP.2013.2286917","volume":"22","author":"H Doi","year":"2014","unstructured":"Doi, H., Toda, T., Nakamura, K., Saruwatari, H., & Shikano, K. (2014). Alaryngeal speech enhancement based on one-to-many eigenvoice conversion. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 22(1), 172\u2013183.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"9643_CR16","unstructured":"Gibiansky, A., Arik, S., Diamos, G., Miller, J., Peng, K., Ping, W., et al. (2017). Deep voice 2: Multi-speaker neural text-to-speech. Advances in Neural Information Processing Systems, 2962\u20132970."},{"issue":"2","key":"9643_CR17","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., & Lim, J. (1984). Signal estimation from modified short-time fourier transform. IEEE Transactions on Acoustics, Speech, and Signal Processing, 32(2), 236\u2013243.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"3","key":"9643_CR18","doi-asserted-by":"publisher","first-page":"806","DOI":"10.1109\/TASL.2011.2165944","volume":"20","author":"E Helander","year":"2012","unstructured":"Helander, E., Sil\u00e9n, H., Virtanen, T., & Gabbouj, M. (2012). Voice conversion using dynamic kernel partial least squares regression. IEEE Transactions on Audio, Speech, and Language Processing, 20(3), 806\u2013817.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"issue":"2","key":"9643_CR19","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1016\/0167-6393(94)00051-B","volume":"16","author":"N Iwahashi","year":"1995","unstructured":"Iwahashi, N., & Sagisaka, Y. (1995). Speech spectrum conversion based on speaker interpolation and multi-functional representation with weighting by radial basis function networks. Speech Communication, 16(2), 139\u2013151.","journal-title":"Speech Communication"},{"key":"9643_CR20","doi-asserted-by":"crossref","unstructured":"Kain, A., & Macon, M.\u00a0W. (1998). Spectral voice conversion for text-to-speech synthesis. In Proceedings of the 1998 IEEE international conference on acoustics, speech and signal processing, 1998. (Vol.\u00a01, pp. 285\u2013288). IEEE.","DOI":"10.1109\/ICASSP.1998.674423"},{"key":"9643_CR21","unstructured":"Kain, A.\u00a0B. (2001). High resolution voice transformation."},{"key":"9643_CR22","doi-asserted-by":"publisher","first-page":"881","DOI":"10.1109\/TPAMI.2002.1017616","volume":"7","author":"T Kanungo","year":"2002","unstructured":"Kanungo, T., Mount, D. M., Netanyahu, N. S., Piatko, C. D., Silverman, R., & Wu, A. Y. (2002). An efficient k-means clustering algorithm: Analysis and implementation. IEEE Transactions on Pattern Analysis & Machine Intelligence, 7, 881\u2013892.","journal-title":"IEEE Transactions on Pattern Analysis & Machine Intelligence"},{"issue":"3\u20134","key":"9643_CR23","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1016\/S0167-6393(98)00085-5","volume":"27","author":"H Kawahara","year":"1999","unstructured":"Kawahara, H., Masuda-Katsuse, I., & De Cheveigne, A. (1999). Restructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds. Speech Communication, 27(3\u20134), 187\u2013207.","journal-title":"Speech Communication"},{"key":"9643_CR24","doi-asserted-by":"crossref","unstructured":"Kawanami, H., Iwami, Y., Toda, T., Saruwatari, H., & Shikano, K. (2003). GMM-based voice conversion applied to emotional speech synthesis. In Eighth European Conference on Speech Communication and Technology.","DOI":"10.21437\/Eurospeech.2003-661"},{"key":"9643_CR25","doi-asserted-by":"crossref","unstructured":"Kobayashi, K., Toda, T., & Nakamura, S. (2016). F0 transformation techniques for statistical voice conversion with direct waveform modification with spectral differential. In 2016 IEEE Spoken Language Technology Workshop (SLT) (pp. 693\u2013700). IEEE.","DOI":"10.1109\/SLT.2016.7846338"},{"key":"9643_CR26","unstructured":"Kominek, J., & Black, A.\u00a0W. (2004). The CMU arctic speech databases. In Fifth ISCA workshop on speech synthesis."},{"issue":"3","key":"9643_CR27","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1109\/MSP.2014.2359987","volume":"32","author":"Z-H Ling","year":"2015","unstructured":"Ling, Z.-H., Kang, S.-Y., Zen, H., Senior, A., Schuster, M., Qian, X.-J., et al. (2015). Deep learning for acoustic modeling in parametric speech generation: A systematic review of existing techniques and future trends. IEEE Signal Processing Magazine, 32(3), 35\u201352.","journal-title":"IEEE Signal Processing Magazine"},{"key":"9643_CR28","doi-asserted-by":"crossref","unstructured":"Liu, L.-J., Chen, L.-H., Ling, Z.-H., & Dai, L.-R. (2015). Spectral conversion using deep neural networks trained with multi-source speakers. In 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 4849\u20134853). IEEE.","DOI":"10.1109\/ICASSP.2015.7178892"},{"key":"9643_CR29","doi-asserted-by":"crossref","unstructured":"Liu, L.-J., Ling, Z.-H., Jiang, Y., Zhou, M., & Dai, L.-R. (2018). Wavenet vocoder with limited training data for voice conversion. Interspeech, 1983\u20131987.","DOI":"10.21437\/Interspeech.2018-1190"},{"key":"9643_CR30","unstructured":"Lorenzo-Trueba, J., Yamagishi, J., Toda, T., Saito, D., Villavicencio, F., Kinnunen, T., & Ling, Z. (2018). The voice conversion challenge 2018: Promoting development of parallel and nonparallel methods. arXiv:1804.04262 ."},{"key":"9643_CR31","doi-asserted-by":"crossref","unstructured":"Mashimo, M., Toda, T., Kawanami, H., Shikano, K., & Campbell, N. (2002). Cross-language voice conversion evaluation using bilingual databases.","DOI":"10.21437\/ICSLP.2002-138"},{"issue":"2","key":"9643_CR32","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1016\/0167-6393(94)00052-C","volume":"16","author":"H Mizuno","year":"1995","unstructured":"Mizuno, H., & Abe, M. (1995). Voice conversion algorithm based on piecewise linear conversion rules of formant frequency and spectrum tilt. Speech Communication, 16(2), 153\u2013164.","journal-title":"Speech Communication"},{"key":"9643_CR33","doi-asserted-by":"crossref","unstructured":"Mouchtaris, A., Van\u00a0der Spiegel, J., & Mueller, P. (2004). A spectral conversion approach to the iterative wiener filter for speech enhancement. In 2004 IEEE international conference on multimedia and expo (ICME)(IEEE Cat. No. 04TH8763) (Vol.\u00a03, pp. 1971\u20131974). IEEE.","DOI":"10.1109\/ICME.2004.1394648"},{"key":"9643_CR34","unstructured":"Nair, V., & Hinton, G.\u00a0E. (2010). Rectified linear units improve restricted Boltzmann machines. In Proceedings of the 27th international conference on machine learning (ICML-10) (pp. 807\u2013814)."},{"issue":"1","key":"9643_CR35","doi-asserted-by":"publisher","first-page":"134","DOI":"10.1016\/j.specom.2011.07.007","volume":"54","author":"K Nakamura","year":"2012","unstructured":"Nakamura, K., Toda, T., Saruwatari, H., & Shikano, K. (2012). Speaking-aid systems using GMM-based voice conversion for electrolaryngeal speech. Speech Communication, 54(1), 134\u2013146.","journal-title":"Speech Communication"},{"key":"9643_CR36","doi-asserted-by":"crossref","unstructured":"Nakashika, T., Takashima, R., Takiguchi, T., & Ariki, Y. (2013). Voice conversion in high-order eigen space using deep belief nets. Interspeech, 369\u2013372.","DOI":"10.21437\/Interspeech.2013-102"},{"key":"9643_CR37","doi-asserted-by":"crossref","unstructured":"Nakashika, T., Takiguchi, T., & Ariki, Y. (2014). High-order sequence modeling using speaker-dependent recurrent temporal restricted Boltzmann machines for voice conversion. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-447"},{"key":"9643_CR38","unstructured":"Oord, A. v.\u00a0d., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A., & Kavukcuoglu, K. (2016). Wavenet: A generative model for raw audio. arXiv:1609.03499 ."},{"issue":"2","key":"9643_CR39","doi-asserted-by":"publisher","first-page":"458","DOI":"10.1121\/1.1911395","volume":"45","author":"AV Oppenheim","year":"1969","unstructured":"Oppenheim, A. V. (1969). Speech analysis-synthesis system based on homomorphic filtering. The Journal of the Acoustical Society of America, 45(2), 458\u2013465.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"9643_CR40","doi-asserted-by":"crossref","unstructured":"Orphanidou, C., Moroz, I.\u00a0M., & Roberts, S.\u00a0J. (2007). Multiscale voice morphing using radial basis function analysis. In Algorithms for Approximation (pp. 61\u201369). Springer, Berlin.","DOI":"10.1007\/978-3-540-46551-5_5"},{"key":"9643_CR41","doi-asserted-by":"crossref","unstructured":"Park, K.-Y., & Kim, H.\u00a0S. (2000). Narrowband to wideband conversion of speech using GMM based transformation. In 2000 IEEE international conference on acoustics, speech, and signal processing. Proceedings (Cat. No. 00CH37100) (Vol.\u00a03, pp. 1843\u20131846). IEEE.","DOI":"10.1109\/ICASSP.2000.862114"},{"key":"9643_CR42","doi-asserted-by":"crossref","unstructured":"Ramani, B., Jeeva, M.\u00a0A., Vijayalakshmi, P., & Nagarajan, T. (2014). Cross-lingual voice conversion-based polyglot speech synthesizer for Indian languages. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-179"},{"issue":"3","key":"9643_CR43","first-page":"1","volume":"5","author":"DE Rumelhart","year":"1988","unstructured":"Rumelhart, D. E., Hinton, G. E., Williams, R. J., et al. (1988). Learning representations by back-propagating errors. Cognitive Modeling, 5(3), 1.","journal-title":"Cognitive Modeling"},{"issue":"8","key":"9643_CR44","doi-asserted-by":"publisher","first-page":"1925","DOI":"10.1587\/transinf.2017EDL8034","volume":"100","author":"Y Saito","year":"2017","unstructured":"Saito, Y., Takamichi, S., & Saruwatari, H. (2017). Voice conversion using input-to-output highway networks. IEICE Transactions on Information and Systems, 100(8), 1925\u20131928.","journal-title":"IEICE Transactions on Information and Systems"},{"issue":"1","key":"9643_CR45","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/TASSP.1978.1163055","volume":"26","author":"H Sakoe","year":"1978","unstructured":"Sakoe, H., & Chiba, S. (1978). Dynamic programming algorithm optimization for spoken word recognition. IEEE Transactions on Acoustics, Speech, and Signal Processing, 26(1), 43\u201349.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9643_CR46","first-page":"164","volume":"2","author":"Y Sekii","year":"2017","unstructured":"Sekii, Y., Orihara, R., Kojima, K., Sei, Y., Tahara, Y., & Ohsuga, A. (2017). Fast many-to-one voice conversion using autoencoders. ICAART, 2, 164\u2013174.","journal-title":"ICAART"},{"key":"9643_CR47","doi-asserted-by":"crossref","unstructured":"Seltzer, M.\u00a0L., Acero, A., & Droppo, J. (2005). Robust bandwidth extension of noise-corrupted narrowband speech. In Ninth European conference on speech communication and technology.","DOI":"10.21437\/Interspeech.2005-529"},{"key":"9643_CR48","doi-asserted-by":"crossref","unstructured":"Song, P., Jin, Y., Zheng, W., & Zhao, L. (2014). Text-independent voice conversion using speaker model alignment method from non-parallel speech. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-187"},{"issue":"1","key":"9643_CR49","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1109\/89.890068","volume":"9","author":"Y Stylianou","year":"2001","unstructured":"Stylianou, Y. (2001). Applying the harmonic plus noise model in concatenative speech synthesis. IEEE Transactions on Speech and Audio Processing, 9(1), 21\u201329.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"issue":"2","key":"9643_CR50","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1109\/89.661472","volume":"6","author":"Y Stylianou","year":"1998","unstructured":"Stylianou, Y., Capp\u00e9, O., & Moulines, E. (1998). Continuous probabilistic transform for voice conversion. IEEE Transactions on Speech and Audio Processing, 6(2), 131\u2013142.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9643_CR51","doi-asserted-by":"crossref","unstructured":"Sun, L., Kang, S., Li, K., & Meng, H. (2015). Voice conversion using deep bidirectional long short-term memory based recurrent neural networks. In 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 4869\u20134873). IEEE.","DOI":"10.1109\/ICASSP.2015.7178896"},{"key":"9643_CR52","doi-asserted-by":"crossref","unstructured":"Sundermann, D., Ney, H., & Hoge, H. (2003). VTLN-based cross-language voice conversion. In 2003 IEEE workshop on automatic speech recognition and understanding (IEEE Cat. No. 03EX721) (pp. 676\u2013681). IEEE.","DOI":"10.1109\/ASRU.2003.1318521"},{"key":"9643_CR53","doi-asserted-by":"crossref","unstructured":"Tamamori, A., Hayashi, T., Kobayashi, K., Takeda, K., & Toda, T. (2017). Speaker-dependent wavenet vocoder. Interspeech, 1118\u20131122.","DOI":"10.21437\/Interspeech.2017-314"},{"issue":"8","key":"9643_CR54","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TASL.2007.907344","volume":"15","author":"T Toda","year":"2007","unstructured":"Toda, T., Black, A. W., & Tokuda, K. (2007). Voice conversion based on maximum-likelihood estimation of spectral parameter trajectory. IEEE Transactions on Audio, Speech, and Language Processing, 15(8), 2222\u20132235.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9643_CR55","doi-asserted-by":"crossref","unstructured":"Toda, T., Chen, L.-H., Saito, D., Villavicencio, F., Wester, M., Wu, Z., et al. (2016). The voice conversion challenge 2016. Interspeech, 1632\u20131636.","DOI":"10.21437\/Interspeech.2016-1066"},{"issue":"5","key":"9643_CR56","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/TASL.2010.2041113","volume":"18","author":"O Turk","year":"2010","unstructured":"Turk, O., & Schroder, M. (2010). Evaluation of expressive speech synthesis with voice conversion and copy resynthesis techniques. IEEE Transactions on Audio, Speech, and Language Processing, 18(5), 965\u2013973.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9643_CR57","unstructured":"Upperman, G. (2004). Linear predictive coding in voice conversion."},{"issue":"2\u20133","key":"9643_CR58","doi-asserted-by":"publisher","first-page":"175","DOI":"10.1016\/0167-6393(92)90012-V","volume":"11","author":"H Valbret","year":"1992","unstructured":"Valbret, H., Moulines, E., & Tubach, J.-P. (1992). Voice transformation using Psola technique. Speech Communication, 11(2\u20133), 175\u2013187.","journal-title":"Speech Communication"},{"key":"9643_CR59","doi-asserted-by":"crossref","unstructured":"Verhelst, W., & Mertens, J. (1996). Voice conversion using partitions of spectral feature space. In 1996 IEEE international conference on acoustics, speech, and signal processing conference proceedings (Vol.\u00a01, pp. 365\u2013368). IEEE.","DOI":"10.1109\/ICASSP.1996.541108"},{"key":"9643_CR60","doi-asserted-by":"crossref","unstructured":"Villavicencio, F., & Bonada, J. (2010). Applying voice conversion to concatenative singing-voice synthesis. In Eleventh annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2010-596"},{"key":"9643_CR61","doi-asserted-by":"crossref","unstructured":"Watanabe, T., Murakami, T., Namba, M., Hoya, T., & Ishida, Y. (2002). Transformation of spectral envelope for voice conversion based on radial basis function networks. In Seventh international conference on spoken language processing.","DOI":"10.21437\/ICSLP.2002-136"},{"key":"9643_CR62","doi-asserted-by":"crossref","unstructured":"Werghi, A., Di\u00a0Martino, J., & Jebara, S.\u00a0B. (2010). On the use of an iterative estimation of continuous probabilistic transforms for voice conversion. In 2010 5th international symposium on I\/V communications and mobile network (pp. 1\u20134). IEEE.","DOI":"10.1109\/ISVC.2010.5656149"},{"key":"9643_CR63","doi-asserted-by":"crossref","unstructured":"Wester, M., Wu, Z., & Yamagishi, J. (2016). Analysis of the voice conversion challenge 2016 evaluation results. Interspeech, 1637\u20131641.","DOI":"10.21437\/Interspeech.2016-1331"},{"key":"9643_CR64","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1016\/j.specom.2013.11.005","volume":"58","author":"N Xu","year":"2014","unstructured":"Xu, N., Tang, Y., Bao, J., Jiang, A., Liu, X., & Yang, Z. (2014). Voice conversion based on gaussian processes by coherent and asymmetric training with limited training data. Speech Communication, 58, 124\u2013138.","journal-title":"Speech Communication"},{"issue":"1","key":"9643_CR65","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1109\/MSP.2010.939038","volume":"28","author":"D Yu","year":"2010","unstructured":"Yu, D., & Deng, L. (2010). Deep learning and its applications to signal and information processing [exploratory DSP]. IEEE Signal Processing Magazine, 28(1), 145\u2013154.","journal-title":"IEEE Signal Processing Magazine"},{"key":"9643_CR66","volume-title":"Automatic Speech Recognition","author":"D Yu","year":"2016","unstructured":"Yu, D., & Deng, L. (2016). Automatic Speech Recognition. Berlin: Springer."},{"key":"9643_CR67","doi-asserted-by":"crossref","unstructured":"Zhu, X., Beauregard, G.\u00a0T., & Wyse, L. (2006). Real-time iterative spectrum inversion with look-ahead. In 2006 IEEE international conference on multimedia and expo (pp. 229\u2013232). IEEE.","DOI":"10.1109\/ICME.2006.262424"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-019-09643-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-019-09643-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-019-09643-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,24]],"date-time":"2024-07-24T19:00:33Z","timestamp":1721847633000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-019-09643-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,8]]},"references-count":67,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["9643"],"URL":"https:\/\/doi.org\/10.1007\/s10772-019-09643-4","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2019,10,8]]},"assertion":[{"value":"24 April 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 October 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}