{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T04:32:42Z","timestamp":1787632362281,"version":"build-2736575974"},"reference-count":85,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2018,9,1]],"date-time":"2018-09-01T00:00:00Z","timestamp":1535760000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/OAPA.html"}],"funder":[{"DOI":"10.13039\/501100004663","name":"Ministry of Science and Technology, Taiwan","doi-asserted-by":"publisher","award":["MOST 106-3114-E-011-004-"],"award-info":[{"award-number":["MOST 106-3114-E-011-004-"]}],"id":[{"id":"10.13039\/501100004663","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2018,9]]},"DOI":"10.1109\/taslp.2018.2821903","type":"journal-article","created":{"date-parts":[[2018,4,5]],"date-time":"2018-04-05T21:18:14Z","timestamp":1522963094000},"page":"1570-1584","source":"Crossref","is-referenced-by-count":253,"title":["End-to-End Waveform Utterance Enhancement for Direct Evaluation Metrics Optimization by Fully Convolutional Neural Networks"],"prefix":"10.1109","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3487-8212","authenticated-orcid":false,"given":"Szu-Wei","family":"Fu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao-Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6956-0418","authenticated-orcid":false,"given":"Yu","family":"Tsao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xugang","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hisashi","family":"Kawai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1159\/000113510"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1044\/1092-4388(2002\/046)"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2017.8282144"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/TBME.2016.2613960"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2004.1316528"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178819"},{"key":"ref74","article-title":"Deep neural network approach for single channel speech enhancement processing","author":"li","year":"2016"},{"key":"ref39","first-page":"1","article-title":"Nondifferentiable optimization via approximation","author":"bertsekas","year":"1975","journal-title":"Nondifferentiable Optimization"},{"key":"ref75","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1126\/science.270.5234.303","article-title":"Speech recognition with primarily temporal cues","volume":"270","author":"shannon","year":"1995","journal-title":"Science"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref78","first-page":"3576","article-title":"Integration of DNN based speech enhancement and ASR","author":"astudillo","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1049\/ip-vis:19960758"},{"key":"ref33","article-title":"Perceptual evaluation of speech quality (PESQ),\n an objective method for end-to-end speech quality assessment of narrowband telephone networks and speech codecs","author":"rix","year":"2001","journal-title":"International Telecommunication Union ITU-T Recommendation"},{"key":"ref32","author":"benesty","year":"2005","journal-title":"Speech Enhancement"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164453"},{"key":"ref30","article-title":"Multi-resolution fully convolutional\n neural networks for monaural audio source separation","author":"grais","year":"2017"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2017.8281993"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2114881"},{"key":"ref60","article-title":"DARPA TIMIT acoustic-phonetic continuous speech corpus","author":"lyons","year":"1993","journal-title":"Nat Inst Stand Technol"},{"key":"ref62","first-page":"126","article-title":"The second &#x2018;CHiME'speech separation\n and recognition challenge: Datasets, tasks and baselines","author":"vincent","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1097\/AUD.0b013e31803154d0"},{"key":"ref63","article-title":"100 Nonspeech Environmental Sounds","author":"hu","year":"2004"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/HSCMA.2017.7895577"},{"key":"ref64","article-title":"Rectifier\n nonlinearities improve neural network acoustic models","author":"maas","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-64680-0_7"},{"key":"ref65","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014"},{"key":"ref66","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"0","journal-title":"Proceedings of the 32nd Intl Conf on Machine Learning"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2716443"},{"key":"ref67","article-title":"RMSProp: Divide the gradient by a running average of its recent magnitude","author":"hinton","year":"2012","journal-title":"Coursera Neural networks for machine learning lecture 6 5"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-49127-9_43"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2008.09.001"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ChinaSIP.2014.6889204"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2013.2291240"},{"key":"ref20","first-page":"2008","article-title":"Conditional generative adversarial networks for speech enhancement and noise-robust speaker\n verification","author":"michelsanti","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2628641"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2468583"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/GlobalSIP.2014.7032183"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/GlobalSIP.2017.8309164"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178061"},{"key":"ref25","first-page":"3274","article-title":"Speech enhancement and\n recognition using multi-task learning of long short-term memory recurrent neural networks","author":"chen","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref50","article-title":"A Wavenet\n for speech denoising","author":"rethage","year":"2017"},{"key":"ref51","author":"oppenheim","year":"1999","journal-title":"Discrete-Time Signal Processing"},{"key":"ref59","article-title":"Theano: A Python framework for fast\n computation of mathematical expressions","author":"team","year":"2016"},{"key":"ref58","article-title":"Keras","author":"chollet","year":"2015"},{"key":"ref57","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1007\/3-540-49430-8_3","article-title":"Early stopping&#x2014;But when?","author":"prechelt","year":"1998","journal-title":"Neural Networks Tricks of the Trade"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2714424"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2365594"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2013.11.003"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1016\/j.disopt.2004.03.007"},{"key":"ref52","article-title":"Wavenet: A generative\n model for raw audio","year":"2016"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.02.001"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1284"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952122"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472673"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2512042"},{"key":"ref14","article-title":"Complex spectrogram enhancement by\n convolutional neural network with multi-metrics learning","author":"fu","year":"0","journal-title":"Proc IEEE 27th Int Workshop Mach Learn Signal Process"},{"key":"ref15","first-page":"1508","article-title":"Multi-objective learning and mask-based post-processing for\n deep neural network based speech enhancement","author":"xu","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref82","article-title":"Does\n speech enhancement work with end-to-end ASR objectives?: Experimental analysis of multichannel end-to-end ASR","author":"ochiai","year":"0","journal-title":"Proc Mach Learn Signal Process"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-671"},{"key":"ref81","first-page":"2632","article-title":"Multichannel end-to-end speech\n recognition","author":"ochiai","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2015.7415295"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2369251"},{"key":"ref18","article-title":"Supervised speech separation based on deep learning: An overview","author":"wang","year":"2017"},{"key":"ref83","article-title":"Speech Recognition (Version 3.6) [Software]","author":"zhang","year":"2017"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953223"},{"key":"ref80","article-title":"Exploring speech enhancement with generative adversarial networks for robust speech recognition","author":"donahue","year":"2017"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2364452"},{"key":"ref3","first-page":"2670","article-title":"Dynamic noise aware training for speech enhancement\n based on deep neural networks","author":"xu","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref6","first-page":"3768","article-title":"SNR-aware\n convolutional neural network modeling for speech enhancement","author":"fu","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref5","first-page":"3713","article-title":"SNR-based progressive learning of deep neural network\n for speech enhancement","author":"gao","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref85","doi-asserted-by":"crossref","DOI":"10.1109\/ICASSP.2018.8462040","article-title":"Monaural\n speech enhancement using deep neural networks by maximizing a short-time objective intelligibility measure","author":"kolb\u00e6k","year":"2018"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2352935"},{"key":"ref7","first-page":"436","article-title":"Speech enhancement based on deep denoising autoencoder","author":"lu","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref49","first-page":"2013","article-title":"Speech enhancement using Bayesian wavenet","author":"qian","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref9","first-page":"91","article-title":"Speech enhancement with LSTM recurrent neural\n networks and its application to noise-robust ASR","author":"weninger","year":"0","journal-title":"Proc Int Conf Latent Variable Anal Signal Separat"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2010.12.003"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/SPED.2011.5940728"},{"key":"ref48","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-1428","article-title":"SEGAN: Speech enhancement generative adversarial network","author":"pascual","year":"2017"},{"key":"ref47","article-title":"Phase-controlled sound transfer based on maximally-inconsistent\n spectrograms","volume":"5","author":"le roux","year":"2011","journal-title":"Signal"},{"key":"ref42","doi-asserted-by":"crossref","DOI":"10.1201\/b14529","author":"loizou","year":"2013","journal-title":"Speech Enhancement Theory and Practice"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2045180"},{"key":"ref44","article-title":"Speech enhancement and noise-robust automatic speech recognition","author":"thomsen","year":"2015"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2016.11.003"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8361959\/08331910.pdf?arnumber=8331910","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,25]],"date-time":"2022-01-25T23:38:15Z","timestamp":1643153895000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8331910\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,9]]},"references-count":85,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2018.2821903","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,9]]}}}