{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,8,6]],"date-time":"2024-08-06T19:41:22Z","timestamp":1722973282984},"reference-count":28,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"6","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2020,6,1]]},"DOI":"10.1587\/transinf.2019edp7166","type":"journal-article","created":{"date-parts":[[2020,5,31]],"date-time":"2020-05-31T22:09:46Z","timestamp":1590962986000},"page":"1395-1405","source":"Crossref","is-referenced-by-count":0,"title":["Tensor Factor Analysis for Arbitrary Speaker Conversion"],"prefix":"10.1587","volume":"E103.D","author":[{"given":"Daisuke","family":"SAITO","sequence":"first","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nobuaki","family":"MINEMATSU","sequence":"additional","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keikichi","family":"HIROSE","sequence":"additional","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"unstructured":"[1] M. Abe, S. Nakamura, K. Shikano, and H. Kuwabara, \u201cVoice conversion through vector quantization,\u201d Proc. ICASSP, pp.655-658, 1988. 10.1109\/icassp.1988.196671","key":"1"},{"doi-asserted-by":"crossref","unstructured":"[2] A. Kain and M.W. Macon, \u201cSpectral voice conversion for text-to-speech synthesis,\u201d Proc. ICASSP, vol.1, pp.285-288, 1998. 10.1109\/icassp.1998.674423","key":"2","DOI":"10.1109\/ICASSP.1998.674423"},{"doi-asserted-by":"crossref","unstructured":"[3] L. Deng, A. Acero, L. Jiang, J. Droppo, and X. Huang, \u201cHigh-performance robust speech recognition using stereo training data,\u201d Proc. ICASSP, pp.301-304, 2001. 10.1109\/icassp.2001.940827","key":"3","DOI":"10.1109\/ICASSP.2001.940827"},{"doi-asserted-by":"crossref","unstructured":"[4] A. Kunikoshi, Y. Qiao, N. Minematsu, and K. Hirose, \u201cSpeech generation from hand gestures based on space mapping,\u201d Proc. INTERSPEECH, pp.308-311, 2009.","key":"4","DOI":"10.21437\/Interspeech.2009-102"},{"doi-asserted-by":"crossref","unstructured":"[5] S. Desai, E.V. Raghavendra, B. Yegnanarayana, A.W. Black, and K. Prahallad, \u201cVoice conversion using artificial neural networks,\u201d Proc. ICASSP, pp.3893-3896, 2009. 10.1109\/icassp.2009.4960478","key":"5","DOI":"10.1109\/ICASSP.2009.4960478"},{"doi-asserted-by":"crossref","unstructured":"[6] Y. Stylianou, O. Cappe, and E. Moulines, \u201cContinuous probabilistic transform for voice conversion,\u201d IEEE Trans. Speech and Audio Processing, vol.6, no.2, pp.131-142, 1998. 10.1109\/89.661472","key":"6","DOI":"10.1109\/89.661472"},{"doi-asserted-by":"publisher","unstructured":"[7] A. Mouchtaris, J.V. der Spiegel, and P. Mueller, \u201cNonparallel training for voice conversion based on a parameter adaptation approach,\u201d IEEE Trans. Audio, Speech, and Language Processing, vol.14, no.3, pp.952-963, 2006. 10.1109\/tsa.2005.857790","key":"7","DOI":"10.1109\/TSA.2005.857790"},{"unstructured":"[8] C.H. Lee and C.H. Wu, \u201cMap-based adaptation for speech conversion using adaptation data selection and non-parallel training,\u201d Proc. INTERSPEECH, pp.2254-2257, 2006.","key":"8"},{"doi-asserted-by":"publisher","unstructured":"[9] D. Saito, S. Watanabe, A. Nakamura, and N. Minematsu, \u201cStatistical voice converison based on noisy channel model,\u201d IEEE Trans. Speech and Audio Processing, vol.20, no.6, pp.1784-1794, 2012. 10.1109\/tasl.2012.2188628","key":"9","DOI":"10.1109\/TASL.2012.2188628"},{"unstructured":"[10] T. Toda, Y. Ohtani, and K. Shikano, \u201cEigenvoice conversion based on Gaussian mixture model,\u201d Proc. INTERSPEECH, pp.2446-2449, 2006.","key":"10"},{"doi-asserted-by":"publisher","unstructured":"[11] R. Kuhn, J-C. Junqua, P. Nguyen, and N. Niedzielski, \u201cRapid speaker adaptation in Eigenvoice space,\u201d IEEE Trans. Speech and Audio Processing, vol.8, no.6, pp.695-707, 2000. 10.1109\/89.876308","key":"11","DOI":"10.1109\/89.876308"},{"doi-asserted-by":"publisher","unstructured":"[12] P. Kenny, P. Ouellet, N. Dehak, V. Gupta, and P. Dumouchel, \u201cA study of interspeaker variability in speaker verification,\u201d IEEE Trans. Audio, Speech, and Language Processing, vol.16, no.5, pp.980-988, 2008. 10.1109\/tasl.2008.925147","key":"12","DOI":"10.1109\/TASL.2008.925147"},{"doi-asserted-by":"publisher","unstructured":"[13] Z. Wu, T. Kinnunnen, E.S. Chng, and H. Li, \u201cMixture of factor analyzers using priors non-parallel speech for voice converison,\u201d IEEE Signal Processing letters, vol.19, no.12, pp.914-917, 2012. 10.1109\/lsp.2012.2225615","key":"13","DOI":"10.1109\/LSP.2012.2225615"},{"doi-asserted-by":"crossref","unstructured":"[14] M.A.O. Vasilescu and D. Terzopoulos, \u201cMutilinear analysis of image ensembles: TensorFaces,\u201d Proc. ECCV, vol.2350, pp.447-460, 2002. 10.1007\/3-540-47969-4_30","key":"14","DOI":"10.1007\/3-540-47969-4_30"},{"doi-asserted-by":"crossref","unstructured":"[15] D. Saito, K. Yamamoto, N. Minematsu, and K. Hirose, \u201cOne-to-many voice conversion based on tensor representation of speaker space,\u201d Proc. INTERSPEECH, pp.653-656, 2011.","key":"15","DOI":"10.21437\/Interspeech.2011-268"},{"unstructured":"[16] Y. Ohtani, T. Toda, H. Saruwatari, and K. Shikano, \u201cSpeaker adaptive training for one-to-many eigenvoice conversion based on Gaussian mixture model,\u201d Proc. INTERSPEECH, pp.1981-1984, 2007.","key":"16"},{"doi-asserted-by":"crossref","unstructured":"[17] Y. Ohtani, T. Toda, H. Saruwatari, and K. Shikano, \u201cNon-parallel training for many-to-many eigenvoice conversion,\u201d Proc. ICASSP, pp.4822-4825, 2010. 10.1109\/icassp.2010.5495139","key":"17","DOI":"10.1109\/ICASSP.2010.5495139"},{"doi-asserted-by":"publisher","unstructured":"[18] T. Hashimoto, D. Saito, and N. Minematsu, \u201cMany-to-Many and Completely Parallel-Data-Free Voice Conversion Based on Eigenspace DNN,\u201d IEEE\/ACM Transaction on Audio, Speech, and Language Processing, vol.27, no.2, pp.332-341, 2019. 10.1109\/taslp.2018.2878949","key":"18","DOI":"10.1109\/TASLP.2018.2878949"},{"doi-asserted-by":"crossref","unstructured":"[19] T. Toda, Y. Ohtani, and K. Shikano, \u201cOne-to-many and many-to-one voice conversion based on eigenvoices,\u201d Proc. ICASSP, vol.IV, pp.693-696, 2007. 10.1109\/icassp.2007.367303","key":"19","DOI":"10.1109\/ICASSP.2007.367303"},{"doi-asserted-by":"publisher","unstructured":"[20] T. Toda, A.W. Black, and K. Tokuda, \u201cVoice conversion based on maximum-likelihood estimation of spectral parameter trajectory,\u201d IEEE Trans. Audio, Speech, and Language Processing, vol.15, no.8, pp.2222-2235, 2007. 10.1109\/tasl.2007.907344","key":"20","DOI":"10.1109\/TASL.2007.907344"},{"doi-asserted-by":"crossref","unstructured":"[21] L. De Lathauwer, B. De Moor and J. Vandewalle, \u201cA multilinear singular value decomposition,\u201d SIAM Journal on Matrix Analysis and Applications, vol.21, no.4, pp.1253-1278, 2000. 10.1137\/s0895479896305696","key":"21","DOI":"10.1137\/S0895479896305696"},{"doi-asserted-by":"crossref","unstructured":"[22] L.R. Tucker, \u201cSome mathematical notes on three-mode factor analysis,\u201d Psychometrika, vol.31, no.3, pp.279-311, 1966. 10.1007\/bf02289464","key":"22","DOI":"10.1007\/BF02289464"},{"doi-asserted-by":"crossref","unstructured":"[23] Y. Jeong, \u201cSpeaker adaptation based on the multilinear decomposition of training speaker models,\u201d Proc. ICASSP, pp.4870-4873, 2010. 10.1109\/icassp.2010.5495117","key":"23","DOI":"10.1109\/ICASSP.2010.5495117"},{"doi-asserted-by":"crossref","unstructured":"[24] T. Anastasakos, J. McDonough, R. Schwarts, and J. Makhoul, \u201cA compact model for speaker adaptive training,\u201d Proc. ICSLP, vol.2, pp.1137-1140, 1996. 10.1109\/icslp.1996.607807","key":"24","DOI":"10.21437\/ICSLP.1996-253"},{"doi-asserted-by":"crossref","unstructured":"[25] D. Saito, N. Minematsu, and K. Hirose, \u201cEffects of speaker adaptive training on tensor-based arbitrary speaker conversion,\u201d Proc. INTERSPEECH, pp.98-101, 2012.","key":"25","DOI":"10.21437\/Interspeech.2012-35"},{"doi-asserted-by":"crossref","unstructured":"[26] A. Kurematsu, K. Takeda, Y. Sagisaka, S. Katagiri, H. Kuwabara, and K. Shikano, \u201cATR Japanese speech database as a tool of speech recognition and synthesis,\u201d Speech Communication, vol.9, no.4, pp.357-363, 1990. 10.1016\/0167-6393(90)90011-w","key":"26","DOI":"10.1016\/0167-6393(90)90011-W"},{"unstructured":"[27] \u201cJNAS: Japanese newspaper article sentences,\u201d http:\/\/www.milab.is.tsukuba.ac.jp\/jnas\/instruct.html","key":"27"},{"doi-asserted-by":"publisher","unstructured":"[28] H. Kawahara, I. Masuda-Katsuse, and A.de Cheveign\u00e9, \u201cRestructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds,\u201d Speech Communication, vol.27, no.3-4, pp.187-207, 1999. 10.1016\/s0167-6393(98)00085-5","key":"28","DOI":"10.1016\/S0167-6393(98)00085-5"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E103.D\/6\/E103.D_2019EDP7166\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,6]],"date-time":"2024-08-06T18:48:06Z","timestamp":1722970086000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E103.D\/6\/E103.D_2019EDP7166\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,6,1]]},"references-count":28,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2020]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2019edp7166","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2020,6,1]]}}}