{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2022,7,9]],"date-time":"2022-07-09T14:40:19Z","timestamp":1657377619314},"reference-count":36,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2016]]},"DOI":"10.1587\/transinf.2016slp0006","type":"journal-article","created":{"date-parts":[[2016,9,30]],"date-time":"2016-09-30T22:23:10Z","timestamp":1475274190000},"page":"2481-2489","source":"Crossref","is-referenced-by-count":1,"title":["Statistical Bandwidth Extension for Speech Synthesis Based on Gaussian Mixture Model with Sub-Band Basis Spectrum Model"],"prefix":"10.1587","volume":"E99.D","author":[{"given":"Yamato","family":"OHTANI","sequence":"first","affiliation":[{"name":"Knowledge Media Laboratory, Corporate Research & Development Center, Toshiba Corporation"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masatsune","family":"TAMURA","sequence":"additional","affiliation":[{"name":"Knowledge Media Laboratory, Corporate Research & Development Center, Toshiba Corporation"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masahiro","family":"MORITA","sequence":"additional","affiliation":[{"name":"Knowledge Media Laboratory, Corporate Research & Development Center, Toshiba Corporation"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masami","family":"AKAMINE","sequence":"additional","affiliation":[{"name":"Knowledge Media Laboratory, Corporate Research & Development Center, Toshiba Corporation"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] B. Iser, W. Minker, and G, Schmidt, Bandwidth extension of speech signals, Lecture Notes in Electrical Engineering, vol.13, Springer US, 2008.","DOI":"10.1007\/978-0-387-68899-2"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] F. Nagel and S. Disch, \u201cA harmonic bandwidth extension method for audio codecs,\u201d Proc. ICASSP, pp.145-148, April 2009.","DOI":"10.1109\/ICASSP.2009.4959541"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] C. Liu, Q.-J. Fu, and S.S. Narayanan, \u201cEffect of bandwidth extension to telephone speech recognition in cochlear implant users,\u201d J. Acoust. Soc. Am., vol.125, no.2, pp.EL77-EL83, Feb. 2009.","DOI":"10.1121\/1.3062145"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] Y. Hu and P.C. Loizou, \u201cEffects of introducing low-frequency harmonics in the perception of vocoded telephone speech,\u201d J. Acoust. Soc. Am., vol.128, no.3, pp.1280-1289, Sept. 2010.","DOI":"10.1121\/1.3463803"},{"key":"5","unstructured":"[5] D. Macho, \u201cNarrowband to wideband feature expansion for robust multilingual ASR,\u201d Proc. INTERSPEECH2007, pp.1118-1121, Aug. 2007."},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] M.L. Seltzer and A. Acero, \u201cTraining wideband acoustic models using mixed-bandwidth training data via feature bandwidth extension,\u201d Proc. ICASSP, vol.1, pp.921-924, March 2005.","DOI":"10.1109\/ICASSP.2005.1415265"},{"key":"7","unstructured":"[7] M. Seltzer and A. Acero, \u201cDNN-based speech bandwidth expansion and its application to adding high-frequency missing features for automatic speech recognition of narrowband speech,\u201d Proc. INTERSPEECH, pp.2578-2582, Sept. 2015."},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] J. Yamagishi, C. Veaux, S. King, and S. Renals, \u201cSpeech synthesis technologies for individuals with vocal disabilities: Voice banking and reconstruction,\u201d Acoust. Sci. Tech., vol.33, no.1, pp.1-5, Jan. 2012.","DOI":"10.1250\/ast.33.1"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] K. Nakamura, T. Toda, H. Saruwatari, and K. Shikano, \u201cSpeaking-aid systems using GMM-based voice conversion for electrolaryngeal speech,\u201d Speech Commun., vol.54, no.1, pp.134-146, Jan. 2012.","DOI":"10.1016\/j.specom.2011.07.007"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] A. Waibel, A.N. Jain, A.E. McNair, H. Saito, A.G. Hauptmann, and J. Tebelskis, \u201cJANUS: A speech-to-speech translation system using connectionist and symbolic processing strategies,\u201d Proc. ICASSP, vol.2, pp.793-796, April 1992.","DOI":"10.1109\/ICASSP.1991.150456"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] S. Matsuda, X. Hu, Y. Shiga, H. Kashioka, C. Hori, K. Yasuda, H. Okuma, M. Uchiyama, E. Sumita, H. Kawai, and S. Nakamura, \u201cMultilingual speech-to-speech translation system: VoiceTra,\u201d 2013 IEEE 14th International Conference on Mobile Data Management (MDM), vol.2, pp.229-233, June 2013.","DOI":"10.1109\/MDM.2013.99"},{"key":"12","unstructured":"[12] Y. Yoshida and M. Abe, \u201cAn algorithm to reconstruct wideband speech form narrowband speech based on codebook mapping,\u201d Proc. ICSLP, pp.1591-1594, Sept. 1994."},{"key":"13","unstructured":"[13] H. Carl and U. Heute, \u201cBandwidth enhancement of narrow-band speech signals,\u201d Proc. EUSIPCO, vol.2, pp.1178-1181, Sept. 1994."},{"key":"14","unstructured":"[14] K.-Y. Park and H.S. Kim, \u201cNarrowband to wideband conversion of speech using GMM based transformation,\u201d Proc. ICASSP, vol.3, pp.1843-1846, 2000."},{"key":"15","unstructured":"[15] W. Fujitsuru, H. Sekimoto, T. Toda, H. Saruwatari, and K. Shikano, \u201cBandwidth extension of cellular phone speech based on maximum likelihood estimation with GMM,\u201d 2008 RISP International Workshop on Nonlinear Circuits and Signal Processing, pp.283-286, March 2008."},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] P. Jax and P. Vary, \u201cOn artificial bandwidth extension of telephone speech,\u201d Signal Process., vol.83, no.8, pp.1707-1719, Aug. 2003.","DOI":"10.1016\/S0165-1684(03)00082-3"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] P. Bauer and T. Fingscheidt, \u201cAn HMM-based artificial bandwidth extension evaluated by cross-language training and test,\u201d Proc. ICASSP, pp.4589-4592, April 2008.","DOI":"10.1109\/ICASSP.2008.4518678"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] J. Kontio, L. Laaksonen, and P. Alku, \u201cNeural network-based artificial bandwidth expansion of speech,\u201d IEEE Trans. Audio Speech Language Process., vol.15, no.3, pp.873-881, March 2007.","DOI":"10.1109\/TASL.2006.885934"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] K. Li and C.-H. Lee, \u201cA deep neural network approach to speech bandwidth expansion,\u201d Proc. ICASSP, pp.4395-4399, April 2015.","DOI":"10.1109\/ICASSP.2015.7178801"},{"key":"20","unstructured":"[20] B. Liu, J. Tao, Z. Wen, Y. Li, and D. Bukhari, \u201cA novel method of artificial bandwidth extension using deep architecture,\u201d Proc. INTERSPEECH, pp.2598-2602, Sept. 2015."},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] H. Pulakka, U. Remes, S. Yrttiaho, K. Palomaki, M. Kurimo, and P. Alku, \u201cBandwidth extension of telephone speech to low frequencies using sinusoidal synthesis and a Gaussian mixture model,\u201d IEEE Trans. Audio Speech Language Process., vol.20, no.8, pp.2219-2231, Oct. 2012.","DOI":"10.1109\/TASL.2012.2199110"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] Y. Stylianou, O. Capp\u00e9, and E. Moulines, \u201cContinuous probabilistic transform for voice conversion,\u201d IEEE Trans. Speech Audio Process., vol.6, no.2, pp.131-142, 1998.","DOI":"10.1109\/89.661472"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] A. Kain and M.W. Macon, \u201cSpectral voice conversion for text-to-speech synthesis,\u201d Proc. ICASSP, vol.1, pp.285-288, May 1998.","DOI":"10.1109\/ICASSP.1998.674423"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] T. Toda, A.W. Black, and K. Tokuda, \u201cVoice conversion based on maximum-likelihood estimation of spectral parameter trajectory,\u201d IEEE Trans. Audio Speech Language Process., vol.15, no.8, pp.2222-2235, Nov. 2007.","DOI":"10.1109\/TASL.2007.907344"},{"key":"25","unstructured":"[25] M. Tamura, T. Kagoshima, and M. Akamine, \u201cSub-band basis spectrum model for pitch-synchronous log-spectrum and phase based on approximation of sparse coding,\u201d Proc. INTERSPEECH, pp.2046-2049, Sept. 2010."},{"key":"26","unstructured":"[26] Y. Ohtani, M. Tamura, M. Morita, and M. Akamine, \u201cGMM-based bandwidth extension using sub-band basis spectrum model,\u201d Proc. INTERSPEECH, pp.2489-2493, Sept. 2014."},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] B.A. Olshausen and D.J. Field, \u201cEmergence of simple-cell respective field properties by learning a sparse code for natural images,\u201d Nature, vol.381, no.6583, pp.607-609, 1996.","DOI":"10.1038\/381607a0"},{"key":"28","unstructured":"[28] H. Kawahara, J. Estill, and O. Fujimura, \u201cAperiodicity extraction and control using mixed mode excitation and group delay manipulation for a high quality speech analysis, modification and synthesis system STRAIGHT,\u201d MAVEBA 2001, Sept. 2001."},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] C. Lawson and R. Hanson, Solving least squares problems, SIAM Classics in Applied Mathematics, 1995 (first published by 1974).","DOI":"10.1137\/1.9781611971217"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] J. Latorre, M.J.F. Gales, S. Buchholz, K. Knill, M. Tamura, Y. Ohtani, and M. Akamine, \u201cContinuous F0 in the source-excitation generation for HMM-based TTS: Do we need voiced\/unvoiced classification?,\u201d Proc. ICASSP, pp.4724-4727, May 2011.","DOI":"10.1109\/ICASSP.2011.5947410"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] P.J.B. Jackson and C.H. Shadle, \u201cPitch-scaled estimation of simultaneous voiced and turbulence-noise components in speech,\u201d IEEE Trans. Speech Audio Process., vol.9, no.7, pp.713-726, Sept. 2001.","DOI":"10.1109\/89.952489"},{"key":"32","unstructured":"[32] C.H. Lee and C.H. Wu, \u201cMAP-based adaptation for speech conversion using adaptation data selection and non-parallel training,\u201d Proc. INTERSPEECH, pp.2254-2257, Sept. 2006."},{"key":"33","doi-asserted-by":"crossref","unstructured":"[33] T. Toda, Y. Ohtani, and K. Shikano, \u201cOne-to-many and many-to-one voice conversion based on eigenvoices,\u201d Proc. ICASSP, vol.4, pp.1249-1252, April 2007.","DOI":"10.1109\/ICASSP.2007.367303"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] A. Mouchtaris, J. Van der Spiegel, and P. Mueller, \u201cNonparallel training for voice conversion based on a parameter adaptation approach,\u201d IEEE Trans. Audio Speech Language Process., vol.14, no.3, pp.952-963, May 2006.","DOI":"10.1109\/TSA.2005.857790"},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] Y. Ohtani, T. Toda, H. Saruwatari, and K. Shikano, \u201cAdaptive training for voice conversion based on eigenvoices,\u201d IEICE Trans. Inf. &amp; Syst., vol.E93-D, no.6, pp.1589-1598, June 2010.","DOI":"10.1587\/transinf.E93.D.1589"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] D. Saito, N. Minematsu, and K. Horose, \u201cEffects of speaker adaptive training on tensor-based arbitrary speaker conversion,\u201d Proc. INTERSPEECH, pp.98-101, Sept. 2012.","DOI":"10.21437\/Interspeech.2012-35"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E99.D\/10\/E99.D_2016SLP0006\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,9]],"date-time":"2022-07-09T13:58:44Z","timestamp":1657375124000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E99.D\/10\/E99.D_2016SLP0006\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"references-count":36,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2016]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2016slp0006","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016]]}}}