{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T11:55:25Z","timestamp":1773921325789,"version":"3.50.1"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2018,5,1]],"date-time":"2018-05-01T00:00:00Z","timestamp":1525132800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"name":"National Key Research and Development Project of China","award":["2017YFB1002202"],"award-info":[{"award-number":["2017YFB1002202"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["U1636201"],"award-info":[{"award-number":["U1636201"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2018,5]]},"DOI":"10.1109\/taslp.2018.2798811","type":"journal-article","created":{"date-parts":[[2018,1,26]],"date-time":"2018-01-26T19:17:05Z","timestamp":1516994225000},"page":"883-894","source":"Crossref","is-referenced-by-count":53,"title":["Waveform Modeling and Generation Using Hierarchical Recurrent Neural Networks for Speech Bandwidth Extension"],"prefix":"10.1109","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7853-5273","authenticated-orcid":false,"given":"Zhen-Hua","family":"Ling","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6668-022X","authenticated-orcid":false,"given":"Yang","family":"Ai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li-Rong","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Deep voice: Real-time neural text-to-speech","author":"arik","year":"2017"},{"key":"ref38","doi-asserted-by":"crossref","first-page":"1558","DOI":"10.1109\/PROC.1977.10770","article-title":"a unified approach to short-time fourier analysis and synthesis","volume":"65","author":"allen","year":"1977","journal-title":"Proceedings of the IEEE"},{"key":"ref33","article-title":"WaveNet: A generative model for raw audio","author":"oord","year":"2016"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-678"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2016.2621053"},{"key":"ref30","first-page":"1","article-title":"Restoring high frequency spectral envelopes using neural networks for speech bandwidth extension","author":"gu","year":"2015","journal-title":"Proc Int Joint Conf Neural Netw"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"ref36","doi-asserted-by":"crossref","first-page":"237","DOI":"10.21437\/Interspeech.2011-91","article-title":"Improved bottleneck features using pretrained deep neural networks.","author":"yu","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-336"},{"key":"ref34","article-title":"SampleRNN: An unconditional end-to-end neural audio generation model","author":"mehri","year":"0","journal-title":"Proc ICLR"},{"key":"ref28","first-page":"2593","article-title":"Speech bandwidth expansion based on deep neural networks","author":"wang","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref27","first-page":"2598","article-title":"A novel method of artificial bandwidth extension using deep architecture","author":"liu","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/IWAENC.2016.7602894"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSPCS.2016.7843305"},{"key":"ref1","first-page":"2494","article-title":"A mel-cepstral analysis technique restoring high frequency components from low-sampling-rate speech","author":"nakamura","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"369","DOI":"10.21437\/Interspeech.2013-102","article-title":"Voice conversion in high-order eigen space using deep belief nets.","author":"nakashika","year":"2013","journal-title":"Proc INTERSPEECH"},{"key":"ref22","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","article-title":"A regression approach to speech enhancement based on deep neural networks","volume":"23","author":"xu","year":"2015","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"ref21","doi-asserted-by":"crossref","first-page":"436","DOI":"10.21437\/Interspeech.2013-130","article-title":"Speech enhancement based on deep denoising autoencoder","author":"lu","year":"2013","journal-title":"Proc INTERSPEECH"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.885934"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CESA.2006.313565"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178801"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2118206"},{"key":"ref10","first-page":"2489","article-title":"GMM-based bandwidth extension using sub-band basis spectrum model","author":"ohtani","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref11","first-page":"363","article-title":"Speech wideband extension based on Gaussian mixture model","author":"zhang","year":"2009","journal-title":"Chin J Acoust"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-314"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2009.03.037"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICALIP.2014.7009818"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518678"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1326084"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2359987"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2269291"},{"key":"ref18","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2353991"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.940919"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/IranianCEE.2012.6292541"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1979.1170672"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CCECE.2010.5575180"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947504"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1660786"},{"key":"ref49","first-page":"1118","article-title":"Crowdsourcing preference tests, and how to detect cheating","author":"buchholz","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CSNT.2015.233"},{"key":"ref46","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc ICLR"},{"key":"ref45","article-title":"DARPA TIMIT acoustic-phonetic continuous speech corpus CD-ROM. NIST speech disc 1&#x2013;1.1","volume":"93","author":"garofolo","year":"1993","journal-title":"NASA STI\/Recon Technical Report"},{"key":"ref48","article-title":"Assessing the quality of audio and video components in desktop multimedia conferencing","author":"watson","year":"2001"},{"key":"ref47","article-title":"P. 862.2: Wideband extension to recommendation P. 862 for the assessment of wideband telephone networks and speech codecs","author":"recommendation","year":"2007","journal-title":"Int Telecommun Union"},{"key":"ref42","article-title":"G. 711: Pulse code modulation (PCM) of voice frequencies","author":"recommendation","year":"1988","journal-title":"Int Telecommun Union"},{"key":"ref41","article-title":"The USTC system for blizzard challenge 2017","author":"hu","year":"2017","journal-title":"Proc Blizzard Challenge Workshop"},{"key":"ref44","first-page":"1","article-title":"Char2wav: End-to-end speech synthesis","author":"sotelo","year":"2017","journal-title":"ICLR Workshop Track"},{"key":"ref43","first-page":"1964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks.","author":"fan","year":"2014","journal-title":"Proc INTERSPEECH"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8318423\/08270683.pdf?arnumber=8270683","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,3]],"date-time":"2024-07-03T15:23:29Z","timestamp":1720020209000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/8270683\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,5]]},"references-count":49,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2018.2798811","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,5]]}}}