{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,6]],"date-time":"2025-05-06T02:10:08Z","timestamp":1746497408364,"version":"3.40.4"},"reference-count":31,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"11","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2014]]},"DOI":"10.1587\/transinf.2014edp7116","type":"journal-article","created":{"date-parts":[[2014,11,1]],"date-time":"2014-11-01T00:00:54Z","timestamp":1414800054000},"page":"2872-2880","source":"Crossref","is-referenced-by-count":0,"title":["Cross-Dialectal Voice Conversion with Neural Networks"],"prefix":"10.1587","volume":"E97.D","author":[{"given":"Weixun","family":"GAO","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, Donghua Univeristy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiying","family":"CAO","sequence":"additional","affiliation":[{"name":"College of Computer Science & Technology, Donghua Univeristy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yao","family":"QIAN","sequence":"additional","affiliation":[{"name":"Speech Group of Microsoft Research Asia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] Y. Stylianou, O. Cappe, and E. Moulines, \u201cContinuous probabilistic transform for voice conversion,\u201d IEEE Trans. Speech Audio Process., vol.6, no.2, pp.131-142, 2002.","DOI":"10.1109\/89.661472"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] T. Toda, A.W. Black, and K. Tokuda, \u201cVoice conversion based on maximum likelihood estimation of spectral parameter trajectory,\u201d IEEE Trans. Audio, Speech Language Process., vol.15, no.8, pp.2222-2235, 2007.","DOI":"10.1109\/TASL.2007.907344"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] A. Hunt and A. Black, \u201cUnit selection in a concatenative speech synthesis system using a large speech database,\u201d Proc. ICASSP, pp.373-376, 1996.","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] S. Desai, A.W. Black, B. Yegnanarayana, and K. Prahallad, \u201cSpectral mapping using artificial neural networks for voice conversion,\u201d IEEE Trans. Audio Speech Language Process.,\u201d vol.18, no.5, pp.954-964, 2010.","DOI":"10.1109\/TASL.2010.2047683"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] T. Nakashika, R. Takashima, T. Takiguchi, and Y. Ariki, \u201cVoice conversion in high-order eigen space using deep belief nets,\u201d Proc. Interspeech, pp.369-372, 2013.","DOI":"10.21437\/Interspeech.2013-102"},{"key":"6","unstructured":"[6] L.-H. Chen, Z.-H. Ling, Y. Song, and L.-R. Dai, \u201cJoint spectral distribution modeling using restricted Boltzmann machines for voice conversion,\u201d Proc. Interspeech, pp.3053-3056, 2013."},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] Z.-Z. Wu, E.-S. Chng, and H.-Z. Li, \u201cConditional restricted Boltzmann machine for voice conversion,\u201d Proc. ChinaSIP, pp.104-108, 2013.","DOI":"10.1109\/ChinaSIP.2013.6625307"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] D. Sundermann, H. Hoge, A. Bonafonte, H. Ney, A. Black, and S. Narayanan, \u201cText-independent voice conversion based on unit selection,\u201d Proc. ICASSP, pp.81-84, 2006.","DOI":"10.1109\/ICASSP.2006.1659962"},{"key":"9","unstructured":"[9] W.-X. Gao and Q. Cao, \u201cFrequency warping for speaker adaptation in HMM-based speech synthesis,\u201d Journal of Information Science and Engineering, vol.30, no.4, pp.1149-1166, 2014."},{"key":"10","unstructured":"[10] http:\/\/en.wikipedia.org\/wiki\/Shanghainese"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] D.E. Rumelhart, G.E. Hinton, and R.J. Williams, \u201cLearning representations by back-propagating errors,\u201d Nature, vol.323, no.9, pp.533-536, 1986.","DOI":"10.1038\/323533a0"},{"key":"12","unstructured":"[12] K. Tokuda, T. Kobayashi, T. Masuko, T. Kobayashi, and T. Kitamura, \u201cSpeech parameter generation algorithms for HMM-based speech synthesis,\u201d Proc. ICASSP, pp.1315-1318, 2000."},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] S.P. Rath and S. Umesh, \u201cAcoustic class specific VTLN-warping using regression class trees,\u201d Proc. Interspeech, pp.556-559, 2009.","DOI":"10.21437\/Interspeech.2009-199"},{"key":"14","unstructured":"[14] R. Greisbach, O. Esser, and C. Weinstock, Speaker identification by formant contours. in A. Braun, J.-P. K\u00f6ster, (eds.), Studies in Forensic Phonetics: BEIPHOL 64. Trier: Wissenschaftlicher Verlag Trier, pp.49-55, 1995."},{"key":"15","unstructured":"[15] A. Krogh and J.A. Hertz, \u201cA simple weight decay can improve generalization,\u201d in Advance in Neural Information Processing Systems 4, eds. J.E. Moody, S.J. Hanson, and P.R. Lippmann, pp.950-957, Morgan Kauffmann Publishers, San Mateo CA, 1992."},{"key":"16","unstructured":"[16] D. Erhan, Y. Bengio, A. Courville, P.A. Manzagol, P. Vincent, and S. Bengio, \u201cWhy does unsupervised pre-training help deep learning?,\u201d J. Machine Learning Research, no.11, pp.625-660, 2010."},{"key":"17","unstructured":"[17] D. Yu, L. Deng, and G. Dahl, \u201cRoles of pre-training and fine-tuning in context-dependent DBN-HMMs for real-world speech recognition,\u201d Proc. NIPS Workshop, 2010."},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] T.N. Sainath, B. Kingsbury, B. Ramabhadran, P. Fousek, P. Novak, and A.R. Mohamed, \u201cMaking deep belief networks effective for large vocabulary continuous speech recognition,\u201d Proc. IEEE ASRU, pp.30-35, 2011.","DOI":"10.1109\/ASRU.2011.6163900"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] G.E. Dahl, D. Yu, L. Deng, and A. Acero, \u201cContext-dependent pre-trained deep neural networks for large-vocabulary speech recognition,\u201d IEEE Trans. Audio Speech Language Process., vol.20, no.1, pp.30-42, 2012.","DOI":"10.1109\/TASL.2011.2134090"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] G. Hinton, L. Deng, D. Yu, G. Dahl, A. Mohamed, N. Jaitly, A. Senior, V. Vanhoucke, P. Nguyen, T. Sainath, and B. Kingsbury, \u201cDeep neural networks for acoustic modeling in speech recognition: The shared views of four research groups,\u201d IEEE Signal Process. Mag., vol.29, no.6, pp.82-97, 2012.","DOI":"10.1109\/MSP.2012.2205597"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] F. Seide, G. Li, X. Chen, and D. Yu, \u201cFeature engineering in context-dependent deep neural networks for conversational speech transcription,\u201d Proc. IEEE ASRU, pp.24-29, 2011.","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] G.E. Hinton, S. Osindero, and Y.W. Teh, \u201cA fast learning algorithm for deep belief nets,\u201d Neural Computation, vol.18, no.7, pp.1527-1554, 2006.","DOI":"10.1162\/neco.2006.18.7.1527"},{"key":"23","unstructured":"[23] A.-R. Mohamed, D. Yu, and L. Deng, \u201cInvestigation of full-sequence training of deep belief networks for speech recognition,\u201d Proc. Interspeech, pp.2846-2849, 2010."},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] B. Kingsbury, T. Sainath, and H. Soltau, \u201cScalable minimum bayes risk training of deep neural network acoustic models using distributed hessian-free optimization,\u201d Proc. Interspeech, pp.10-13, 2012.","DOI":"10.21437\/Interspeech.2012-3"},{"key":"25","unstructured":"[25] Y. Wu and R. Wang, \u201cMinimum generation error training for HMMBased speech synthesis,\u201d Proc. ICASSP, pp.89-92, 2006."},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] H. Kawahara, I. Masuda-Katsuse, and A. de Cheveigne, \u201cRestructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds,\u201d Speech Commun., vol.27, no.3-4, pp.187-207, 1999.","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"27","unstructured":"[27] D. Talkin, W. Kleijn, and K. Paliwal, A robust algorithm for pitch tracking (RAPT) in Speech Coding and Synthesis, pp.495-518, Elsevier, 1995."},{"key":"28","unstructured":"[28] C.J. Chen, R.A. Gopinath, M.D. Monkowski, M.A. Picheny, and K. Shen, \u201cNew methods in continuous Mandarin speech recognition,\u201d Proc. EUROSPEECH, pp.1543-1546, 1997."},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] Z.-H. Ling, Y.-J. Wu, Y.-P. Wang, L. Qin, and R.-H. Wang, \u201cUSTC system for blizzard challenge 2006 an improved HMM-based speech synthesis method,\u201d Proc. Blizzard Challenge 2006 Workshop, 2006.","DOI":"10.21437\/Blizzard.2006-6"},{"key":"30","unstructured":"[30] M.-L. Lei, Z-H. Ling, and L.-R. Dai, \u201cMinimum generation error training with weighted Euclidean distance on LSP for HMM-based speech synthesis,\u201d Proc. ICASSP, pp.4230-4233, 2010."},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] R. Laroia, N. Phamdo, and N. Farvardin, \u201cRobust and efficient quantization of speech LSP parameters using structured vector quantizers,\u201d Proc. ICASSP, pp.641-644, 1991.","DOI":"10.1109\/ICASSP.1991.150421"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E97.D\/11\/E97.D_2014EDP7116\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,6]],"date-time":"2025-05-06T01:32:30Z","timestamp":1746495150000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E97.D\/11\/E97.D_2014EDP7116\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"references-count":31,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2014]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2014edp7116","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2014]]}}}