{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,27]],"date-time":"2024-09-27T04:15:18Z","timestamp":1727410518417},"reference-count":46,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"6","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2022,6,1]]},"DOI":"10.1587\/transinf.2021edp7234","type":"journal-article","created":{"date-parts":[[2022,5,31]],"date-time":"2022-05-31T22:14:00Z","timestamp":1654035240000},"page":"1196-1210","source":"Crossref","is-referenced-by-count":1,"title":["INmfCA Algorithm for Training of Nonparallel Voice Conversion Systems Based on Non-Negative Matrix Factorization"],"prefix":"10.1587","volume":"E105.D","author":[{"given":"Hitoshi","family":"SUDA","sequence":"first","affiliation":[{"name":"Graduate School of Engineering, The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaku","family":"KOTANI","sequence":"additional","affiliation":[{"name":"Graduate School of Engineering, The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daisuke","family":"SAITO","sequence":"additional","affiliation":[{"name":"Graduate School of Engineering, The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] A. Kain and M.W. Macon, \u201cSpectral voice conversion for text-to-speech synthesis,\u201d Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, pp.285-288, May 1998. 10.1109\/icassp.1998.674423","DOI":"10.1109\/ICASSP.1998.674423"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] Y. Stylianou, O. Capp\u00e9, and E. Moulines, \u201cStatistical methods for voice quality transformation,\u201d Proc. EUROSPEECH, pp.447-450, Sept. 1995.","DOI":"10.21437\/Eurospeech.1995-121"},{"key":"3","doi-asserted-by":"publisher","unstructured":"[3] T. Toda, A.W. Black, and K. Tokuda, \u201cVoice conversion based on maximum-likelihood estimation of spectral parameter trajectory,\u201d IEEE Trans. Audio, Speech, Language Process., vol.15, no.8, pp.2222-2235, Nov. 2007. 10.1109\/tasl.2007.907344","DOI":"10.1109\/TASL.2007.907344"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] L.-H. Chen, Z.-H. Ling, Y. Song, and L.-R. Dai, \u201cJoint spectral distribution modeling using restricted Boltzmann machines for voice conversion,\u201d Proc. INTERSPEECH, pp.3052-3056, Aug. 2013. 10.21437\/interspeech.2013-666","DOI":"10.21437\/Interspeech.2013-666"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] T. Nakashika, T. Takiguchi, and Y. Ariki, \u201cSparse nonlinear representation for voice conversion,\u201d Proc. IEEE International Conference on Multimedia and Expo, pp.1-6, June 2015. 10.1109\/icme.2015.7177437","DOI":"10.1109\/ICME.2015.7177437"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] B. Makki, S.A. Seyedsalehi, N. Sadati, and M.N. Hosseini, \u201cVoice conversion using nonlinear principal component analysis,\u201d Proc. IEEE Symposium on Computational Intelligence in Image and Signal Processing, pp.336-339, April 2007. 10.1109\/ciisp.2007.369191","DOI":"10.1109\/CIISP.2007.369191"},{"key":"7","doi-asserted-by":"publisher","unstructured":"[7] S. Desai, A.W. Black, B. Yegnanarayana, and K. Prahallad, \u201cSpectral mapping using artificial neural networks for voice conversion,\u201d IEEE Trans. Audio, Speech, Language Process., vol.18, no.5, pp.954-964, July 2010. 10.1109\/tasl.2010.2047683","DOI":"10.1109\/TASL.2010.2047683"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] L. Sun, S. Kang, K. Li, and H. Meng, \u201cVoice conversion using deep bidirectional long short-term memory based recurrent neural networks,\u201d Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, pp.4869-4873, April 2015. 10.1109\/icassp.2015.7178896","DOI":"10.1109\/ICASSP.2015.7178896"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] R. Takashima, T. Takiguchi, and Y. Ariki, \u201cExemplar-based voice conversion in noisy environment,\u201d Proc. IEEE Spoken Language Technology Workshop, pp.313-317, Dec. 2012. 10.1109\/slt.2012.6424242","DOI":"10.1109\/SLT.2012.6424242"},{"key":"10","doi-asserted-by":"publisher","unstructured":"[10] Z. Wu, T. Virtanen, E.S. Chng, and H. Li, \u201cExemplar-based sparse representation with residual compensation for voice conversion,\u201d IEEE\/ACM Trans. Audio, Speech, Language Process., vol.22, no.10, pp.1506-1521, Oct. 2014. 10.1109\/taslp.2014.2333242","DOI":"10.1109\/TASLP.2014.2333242"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] T. Kaneko and H. Kameoka, \u201cCycleGAN-VC: Non-parallel voice conversion using cycle-consistent adversarial networks,\u201d Proc. the 26th European Signal Processing Conference, pp.2100-2104, Sept. 2018. 10.23919\/eusipco.2018.8553236","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"12","doi-asserted-by":"publisher","unstructured":"[12] D. Erro, A. Moreno, and A. Bonafonte, \u201cINCA algorithm for training voice conversion systems from nonparallel corpora,\u201d IEEE Trans. Audio, Speech, Language Process., vol.18, no.5, pp.944-953, July 2010. 10.1109\/tasl.2009.2038669","DOI":"10.1109\/TASL.2009.2038669"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] C.-C. Hsu, H.-T. Hwang, Y.-C. Wu, Y. Tsao, and H.-M. Wang, \u201cVoice conversion from non-parallel corpora using variational auto-encoder,\u201d Proc. Asia-Pacific Signal and Information Processing Association Annual Summit and Conference, pp.1-6, Dec. 2016. 10.1109\/apsipa.2016.7820786","DOI":"10.1109\/APSIPA.2016.7820786"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] D. Saito, S. Watanabe, A. Nakamura, and N. Minematsu, \u201cStatistical voice conversion based on noisy channel model,\u201d IEEE Trans. Audio, Speech, Language Process., vol.20, no.6, pp.1784-1794, Aug. 2012. 10.1109\/tasl.2012.2188628","DOI":"10.1109\/TASL.2012.2188628"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] H. Suda, G. Kotani, and D. Saito, \u201cNonparallel training of exemplar-based voice conversion system using INCA-based alignment technique,\u201d Proc. INTERSPEECH, pp.4681-4685, Oct. 2020. 10.21437\/interspeech.2020-1232","DOI":"10.21437\/Interspeech.2020-1232"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] H. Benisty, D. Malah, and K. Crammer, \u201cNon-parallel voice conversion using joint optimization of alignment by temporal context and spectral distortion,\u201d Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, pp.7909-7913, May 2014. 10.1109\/icassp.2014.6855140","DOI":"10.1109\/ICASSP.2014.6855140"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] N. Shah and H. Patil, \u201cEffectiveness of dynamic features in INCA and temporal context-INCA,\u201d Proc. INTERSPEECH, pp.711-715, Sept. 2018. 10.21437\/interspeech.2018-1538","DOI":"10.21437\/Interspeech.2018-1538"},{"key":"18","doi-asserted-by":"publisher","unstructured":"[18] A. Mouchtaris, J.V. der Spiegel, and P. Mueller, \u201cNonparallel training for voice conversion based on a parameter adaptation approach,\u201d IEEE Trans. Audio, Speech, Language Process., vol.14, no.3, pp.952-963, May 2006. 10.1109\/tsa.2005.857790","DOI":"10.1109\/TSA.2005.857790"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] T. Toda, Y. Ohtani, and K. Shikano, \u201cEigenvoice conversion based on Gaussian mixture model,\u201d Proc. INTERSPEECH, pp.2446-2449, Sept. 2006. 10.21437\/interspeech.2006-613","DOI":"10.21437\/Interspeech.2006-613"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] D. Saito, K. Yamamoto, N. Minematsu, and K. Hirose, \u201cOne-to-many voice conversion based on tensor representation of speaker space,\u201d Proc. INTERSPEECH, pp.653-656, Aug. 2011. 10.21437\/interspeech.2011-268","DOI":"10.21437\/Interspeech.2011-268"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] N. Dehak, P.J. Kenny, R. Dehak, P. Dumouchel, and P. Ouellet, \u201cFront-end factor analysis for speaker verification,\u201d IEEE Trans. Audio, Speech, Language Process., vol.19, no.4, pp.788-798, May 2011. 10.1109\/tasl.2010.2064307","DOI":"10.1109\/TASL.2010.2064307"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] T. Kinnunen, L. Juvela, P. Alku, and J. Yamagishi, \u201cNon-parallel voice conversion using i-vector PLDA: Towards unifying speaker verification and transformation,\u201d Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, pp.5535-5539, March 2017. 10.1109\/icassp.2017.7953215","DOI":"10.1109\/ICASSP.2017.7953215"},{"key":"23","doi-asserted-by":"publisher","unstructured":"[23] J.-X. Zhang, Z.-H. Ling, and L.-R. Dai, \u201cNon-parallel sequence-to-sequence voice conversion with disentangled linguistic and speaker representations,\u201d IEEE\/ACM Trans. Audio, Speech, Language Process., vol.28, pp.540-552, 2020. 10.1109\/taslp.2019.2960721","DOI":"10.1109\/TASLP.2019.2960721"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] L. Sun, K. Li, H. Wang, S. Kang, and H. Meng, \u201cPhonetic posteriorgrams for many-to-one voice conversion without parallel data training,\u201d Proc. IEEE International Conference on Multimedia and Expo, pp.1-6, July 2016.","DOI":"10.1109\/ICME.2016.7552917"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] T. Kaneko, H. Kameoka, K. Tanaka, and N. Hojo, \u201cCycle-GAN-VC2: Improved CycleGAN-based non-parallel voice conversion,\u201d Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, pp.6820-6824, May 2019. 10.1109\/icassp.2019.8682897","DOI":"10.1109\/ICASSP.2019.8682897"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] T. Kaneko, H. Kameoka, K. Tanaka, and N. Hojo, \u201cCycle-GAN-VC3: Examining and improving CycleGAN-VCs for mel-spectrogram conversion,\u201d Proc. INTERSPEECH, pp.2017-2021, Oct. 2020. 10.21437\/interspeech.2020-2280","DOI":"10.21437\/Interspeech.2020-2280"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] T. Kaneko, H. Kameoka, K. Tanaka, and N. Hojo, \u201cStarGAN-VC2: Rethinking conditional methods for StarGAN-Based voice conversion,\u201d Proc. INTERSPEECH, pp.679-683, 2019. 10.21437\/interspeech.2019-2236","DOI":"10.21437\/Interspeech.2019-2236"},{"key":"28","doi-asserted-by":"publisher","unstructured":"[28] H. Kameoka, T. Kaneko, K. Tanaka, and N. Hojo, \u201cACVAE-VC: Non-parallel voice conversion with auxiliary classifier variational autoencoder,\u201d IEEE\/ACM Trans. Audio, Speech, Language Process., vol.27, no.9, pp.1432-1443, Sept. 2019. 10.1109\/taslp.2019.2917232","DOI":"10.1109\/TASLP.2019.2917232"},{"key":"29","doi-asserted-by":"publisher","unstructured":"[29] D.D. Lee and H.S. Seung, \u201cLearning the parts of objects by non-negative matrix factorization,\u201d Nature, vol.401, no.6755, pp.788-791, Oct. 1999. 10.1038\/44565","DOI":"10.1038\/44565"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] P. Smaragdis and J.C. Brown, \u201cNon-negative matrix factorization for polyphonic music transcription,\u201d Proc. IEEE Workshop on Applications of Signal Processing to Audio and Acoustics, pp.177-180, Oct. 2003. 10.1109\/aspaa.2003.1285860","DOI":"10.1109\/ASPAA.2003.1285860"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] M.N. Schmidt, J. Larsen, and F.-T. Hsiao, \u201cWind noise reduction using non-negative sparse coding,\u201d Proc. IEEE Workshop on Machine Learning for Signal Processing, pp.431-436, Aug. 2007. 10.1109\/mlsp.2007.4414345","DOI":"10.1109\/MLSP.2007.4414345"},{"key":"32","doi-asserted-by":"crossref","unstructured":"[32] P. Smaragdis and B. Raj, \u201cExample-driven bandwidth expansion,\u201d Proc. IEEE Workshop on Applications of Signal Processing to Audio and Acoustics, pp.135-138, Oct. 2007. 10.1109\/aspaa.2007.4393004","DOI":"10.1109\/ASPAA.2007.4393004"},{"key":"33","unstructured":"[33] D.D. Lee and H.S. Seung, \u201cAlgorithms for non-negative matrix factorization,\u201d Proc. the Advances in Neural Information Processing Systems 13, pp.556-562, Dec. 2001."},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] N.J. Shah and H.A. Patil, \u201cOn the convergence of INCA algorithm,\u201d Proc. Asia-Pacific Signal and Information Processing Association Annual Summit and Conference, pp.559-562, Dec. 2017. 10.1109\/apsipa.2017.8282095","DOI":"10.1109\/APSIPA.2017.8282095"},{"key":"35","doi-asserted-by":"publisher","unstructured":"[35] M. Morise, F. Yokomori, and K. Ozawa, \u201cWORLD: A vocoder-based high-quality speech synthesis system for real-time applications,\u201d IEICE Trans. Inf. &amp; Syst., vol.E99-D, no.7, pp.1877-1884, July 2016. 10.1587\/transinf.2015edp7457","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"36","doi-asserted-by":"publisher","unstructured":"[36] M. Morise, \u201cD4C, a band-aperiodicity estimator for high-quality speech synthesis,\u201d Speech Communication, vol.84, pp.57-65, Nov. 2016. 10.1016\/j.specom.2016.09.001","DOI":"10.1016\/j.specom.2016.09.001"},{"key":"37","doi-asserted-by":"publisher","unstructured":"[37] M. Pitz and H. Ney, \u201cVocal tract normalization equals linear transformation in cepstral space,\u201d IEEE Transactions on Speech and Audio Processing, vol.13, no.5, pp.930-944, Sept. 2005. 10.1109\/tsa.2005.848881","DOI":"10.1109\/TSA.2005.848881"},{"key":"38","unstructured":"[38] K. Tokuda, T. Kobayashi, and S. Imai, \u201cRecursion formula for calculation of mel generalized cepstrum coefficients,\u201d IEICE Trans. Fundamentals (Japanese Edition), vol.71, no.1, pp.128-131, Jan. 1988."},{"key":"39","doi-asserted-by":"crossref","unstructured":"[39] G. Kotani, H. Suda, D. Saito, and N. Minematsu, \u201cExperimental investigation on the efficacy of Affine-DTW in the quality of voice conversion,\u201d Proc. Asia-Pacific Signal and Information Processing Association Annual Summit and Conference, pp.119-124, Nov. 2019. 10.1109\/apsipaasc47483.2019.9023107","DOI":"10.1109\/APSIPAASC47483.2019.9023107"},{"key":"40","doi-asserted-by":"crossref","unstructured":"[40] G. Zhou, S. Xie, Z. Yang, J.-M. Yang, and Z. He, \u201cMinimum-volume-constrained nonnegative matrix factorization: Enhanced ability of learning parts,\u201d IEEE Trans. Neural Netw., vol.22, no.10, pp.1626-1637, Oct. 2011. 10.1109\/tnn.2011.2164621","DOI":"10.1109\/TNN.2011.2164621"},{"key":"41","unstructured":"[41] P.O. Hoyer, \u201cNon-negative matrix factorization with sparseness constraints,\u201d Journal of Machine Learning Research, vol.5, pp.1457-1469, Dec. 2004."},{"key":"42","doi-asserted-by":"crossref","unstructured":"[42] A. Cichocki, R. Zdunek, and S.i. Amari, \u201cNew algorithms for non-negative matrix factorization in applications to blind source separation,\u201d Proc. IEEE International Conference on Acoustics Speech and Signal Processing Proceedings, pp.621-624, May 2006. 10.1109\/icassp.2006.1661352","DOI":"10.1109\/ICASSP.2006.1661352"},{"key":"43","unstructured":"[43] A. van den Oord, S. Dieleman, H. Zen, K. Simonyan, O. Vinyals, A. Graves, N. Kalchbrenner, A. Senior, and K. Kavukcuoglu, \u201cWaveNet: A generative model for raw audio,\u201d arXiv:1609.03499 [cs], Sept. 2016."},{"key":"44","unstructured":"[44] N. Kalchbrenner, E. Elsen, K. Simonyan, S. Noury, N. Casagrande, E. Lockhart, F. Stimberg, A. van den Oord, S. Dieleman, and K. Kavukcuoglu, \u201cEfficient Neural Audio Synthesis,\u201d arXiv:1802.08435 [cs, eess], June 2018."},{"key":"45","doi-asserted-by":"crossref","unstructured":"[45] R. Prenger, R. Valle, and B. Catanzaro, \u201cWaveGlow: A Flow-based Generative Network for Speech Synthesis,\u201d arXiv:1811.00002 [cs, eess, stat], Oct. 2018.","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"46","doi-asserted-by":"crossref","unstructured":"[46] K. Tanaka, T. Kaneko, N. Hojo, and H. Kameoka,\u201cWaveCycleGAN: Synthetic-to-natural speech waveform conversion using cycle-consistent adversarial networks,\u201d arXiv:1809.10288 [cs, eess, stat], Sept. 2018.","DOI":"10.1109\/SLT.2018.8639636"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E105.D\/6\/E105.D_2021EDP7234\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,26]],"date-time":"2024-09-26T05:48:30Z","timestamp":1727329710000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E105.D\/6\/E105.D_2021EDP7234\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,1]]},"references-count":46,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2022]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2021edp7234","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2022,6,1]]},"article-number":"2021EDP7234"}}