{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:06:05Z","timestamp":1771697165586,"version":"3.50.1"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319646794","type":"print"},{"value":"9783319646800","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_7","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"165-186","source":"Crossref","is-referenced-by-count":24,"title":["Deep Recurrent Networks for Separation and Recognition of Single-Channel Speech in Nonstationary Background Audio"],"prefix":"10.1007","author":[{"given":"Hakan","family":"Erdogan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John R.","family":"Hershey","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonathan","family":"Le Roux","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"7_CR1","volume-title":"Speech Enhancement","author":"J. Benesty","year":"2005","unstructured":"Benesty, J., Makino, S., Chen, J.: Speech Enhancement. Springer Science & Business Media, New York (2005)"},{"issue":"4","key":"7_CR2","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1109\/97.1001645","volume":"9","author":"I. Cohen","year":"2002","unstructured":"Cohen, I.: Optimal speech enhancement under signal presence uncertainty using log-spectral amplitude estimator. IEEE Signal Process. Lett. 9(4), 113\u2013116 (2002)","journal-title":"IEEE Signal Process. Lett."},{"issue":"1","key":"7_CR3","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1109\/97.988717","volume":"9","author":"I. Cohen","year":"2002","unstructured":"Cohen, I., Berdugo, B.: Noise estimation by minima controlled recursive averaging for robust speech enhancement. IEEE Signal Process. Lett. 9(1), 12\u201315 (2002)","journal-title":"IEEE Signal Process. Lett."},{"issue":"6","key":"7_CR4","doi-asserted-by":"crossref","first-page":"1109","DOI":"10.1109\/TASSP.1984.1164453","volume":"32","author":"Y. Ephraim","year":"1984","unstructured":"Ephraim, Y., Malah, D.: Speech enhancement using a minimum-mean square error short-time spectral amplitude estimator. IEEE Trans. Acoust. Speech Signal Process. 32(6), 1109\u20131121 (1984)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"2","key":"7_CR5","doi-asserted-by":"crossref","first-page":"443","DOI":"10.1109\/TASSP.1985.1164550","volume":"33","author":"Y. Ephraim","year":"1985","unstructured":"Ephraim, Y., Malah, D.: Speech enhancement using a minimum mean-square error log-spectral amplitude estimator. IEEE Trans. Acoust. Speech Signal Process. 33(2), 443\u2013445 (1985)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Erdogan, H., Hershey, J.R., Watanabe, S., Le Roux, J.: Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Brisbane (2015)","DOI":"10.1109\/ICASSP.2015.7178061"},{"issue":"7","key":"7_CR7","doi-asserted-by":"crossref","first-page":"2067","DOI":"10.1109\/TASL.2011.2112350","volume":"19","author":"J.F. Gemmeke","year":"2011","unstructured":"Gemmeke, J.F., Virtanen, T., Hurmalainen, A.: Exemplar-based sparse representations for noise robust automatic speech recognition. IEEE Trans. Audio Speech Lang. Process. 19(7), 2067\u20132080 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Grais, E.M., Erdogan, H.: Single channel speech music separation using nonnegative matrix factorization and spectral masks. In: Proceedings of the International Conference on Digital Signal Processing (DSP), pp.\u00a01\u20136 (2011)","DOI":"10.1109\/ICDSP.2011.6004924"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Grais, E.M., Sen, M.U., Erdogan, H.: Deep neural networks for single channel source separation. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Florence (2014)","DOI":"10.1109\/ICASSP.2014.6854299"},{"issue":"8","key":"7_CR10","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S. Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"issue":"1","key":"7_CR11","doi-asserted-by":"crossref","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y. Hu","year":"2008","unstructured":"Hu, Y., Loizou, P.C.: Evaluation of objective quality measures for speech enhancement. IEEE Trans. Audio Speech Lang. Process. 16(1), 229\u2013238 (2008)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Huang, P.S., Kim, M., Hasegawa-Johnson, M., Smaragdis, P.: Deep learning for monaural speech separation. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Florence, pp.\u00a01581\u20131585 (2014)","DOI":"10.1109\/ICASSP.2014.6853860"},{"key":"7_CR13","first-page":"125","volume":"1","author":"R. Kubichek","year":"1993","unstructured":"Kubichek, R.: Mel-cepstral distance measure for objective speech quality assessment. In: IEEE Pacific Rim Conference on Communications, Computers and Signal Processing, vol.\u00a01, pp.\u00a0125\u2013128 (1993)","journal-title":"In: IEEE Pacific Rim Conference on Communications, Computers and Signal Processing"},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Le\u00a0Roux, J., Vincent, E., Mizuno, Y., Kameoka, H., Ono, N., Sagayama, S.: Consistent Wiener filtering: generalized time\u2013frequency masking respecting spectrogram consistency. In: Proceedings of the International Conference on Latent Variable Analysis and Signal Separation, pp.\u00a089\u201396 (2010)","DOI":"10.1007\/978-3-642-15995-4_12"},{"key":"7_CR15","volume-title":"Sparse NMF \u2013 half-baked or well done? Technical Report, TR2015\u2013023, Mitsubishi Electric Research Laboratories (MERL)","author":"J. Roux Le","year":"2015","unstructured":"Le\u00a0Roux, J., Weninger, F.J., Hershey, J.R.: Sparse NMF \u2013 half-baked or well done? Technical Report, TR2015-023, Mitsubishi Electric Research Laboratories (MERL), Cambridge, MA (2015)"},{"key":"7_CR16","unstructured":"Lee, D.D., Seung, H.S.: Algorithms for non-negative matrix factorization. In: Advances in Neural Information Processing Systems (NIPS), pp.\u00a0556\u2013562 (2001)"},{"key":"7_CR17","doi-asserted-by":"crossref","DOI":"10.1201\/b14529","volume-title":"Speech Enhancement: Theory and Practice","author":"P.C. Loizou","year":"2013","unstructured":"Loizou, P.C.: Speech Enhancement: Theory and Practice. CRC Press, Boca Raton, FL (2013)"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Lu, X., Tsao, Y., Matsuda, S., Hori, C.: Speech enhancement based on deep denoising autoencoder. In: Proceedings of the Interspeech, Lyon, pp.\u00a03444\u20133448 (2013)","DOI":"10.21437\/Interspeech.2013-130"},{"key":"7_CR19","unstructured":"Maas, A.L., O\u2019Neil, T.M., Hannun, A.Y., Ng, A.Y.: Recurrent neural network feature enhancement: the 2nd CHiME challenge. In: Proceedings of the CHiME Workshop on Machine Listening in Multisource Environments, Vancouver, pp.\u00a079\u201380 (2013)"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Narayanan, A., Wang, D.: Ideal ratio mask estimation using deep neural networks for robust speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Vancouver, pp.\u00a07092\u20137096 (2013)","DOI":"10.1109\/ICASSP.2013.6639038"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Paul, D.B., Baker, J.M.: The design for the Wall Street Journal-based CSR corpus. In: Proceedings of the Workshop on Speech and Natural Language, pp.\u00a0357\u2013362 (1992)","DOI":"10.3115\/1075527.1075614"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Schmidt, M.N., Olsson, R.K.: Single-channel speech separation using sparse non-negative matrix factorization. In: Proceedings of the Interspeech, Pittsburgh, PA, pp.\u00a01652\u201355 (2006)","DOI":"10.21437\/Interspeech.2006-655"},{"issue":"1","key":"7_CR23","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TASL.2006.876726","volume":"15","author":"P. Smaragdis","year":"2007","unstructured":"Smaragdis, P.: Convolutive speech bases and their application to supervised speech separation. IEEE Trans. Audio Speech Lang. Process. 15(1), 1\u201314 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"7","key":"7_CR24","doi-asserted-by":"crossref","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","volume":"19","author":"C.H. Taal","year":"2011","unstructured":"Taal, C.H., Hendriks, R.C., Heusdens, R., Jensen, J.: An algorithm for intelligibility prediction of time\u2013frequency weighted noisy speech. IEEE Trans. Audio Speech Lang. Process. 19(7), 2125\u20132136 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"4","key":"7_CR25","doi-asserted-by":"crossref","first-page":"1462","DOI":"10.1109\/TSA.2005.858005","volume":"14","author":"E. Vincent","year":"2006","unstructured":"Vincent, E., Gribonval, R., F\u00e9votte, C.: Performance measurement in blind audio source separation. IEEE Trans. Audio Speech Lang. Process. 14(4), 1462\u20131469 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Vincent, E., Barker, J., Watanabe, S., Le Roux, J., Nesta, F., Matassoni, M.: The second \u201cCHiME\u201d speech separation and recognition challenge: datasets, tasks and baselines. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), Vancouver, pp.\u00a0126\u2013130 (2013)","DOI":"10.1109\/ICASSP.2013.6637622"},{"issue":"3","key":"7_CR27","doi-asserted-by":"crossref","first-page":"1066","DOI":"10.1109\/TASL.2006.885253","volume":"15","author":"T. Virtanen","year":"2007","unstructured":"Virtanen, T.: Monaural sound source separation by nonnegative matrix factorization with temporal continuity and sparseness criteria. IEEE Trans. Audio Speech Lang. Process. 15(3), 1066\u20131074 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"4","key":"7_CR28","doi-asserted-by":"crossref","first-page":"796","DOI":"10.1109\/TASLP.2016.2528171","volume":"24","author":"Z.Q. Wang","year":"2016","unstructured":"Wang, Z.Q., Wang, D.: A joint training framework for robust automatic speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(4), 796\u2013806 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"12","key":"7_CR29","doi-asserted-by":"crossref","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","volume":"22","author":"Y. Wang","year":"2014","unstructured":"Wang, Y., Narayanan, A., Wang, D.: On training targets for supervised speech separation. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(12), 1849\u201358 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"7_CR30","unstructured":"Weninger, F., Geiger, J., W\u00f6llmer, M., Schuller, B., Rigoll, G.: The Munich feature enhancement approach to the 2013 CHiME challenge using BLSTM recurrent neural networks. In: Proceedings of the 2nd CHiME Speech Separation and Recognition Challenge held in conjunction with ICASSP 2013, Vancouver, pp.\u00a086\u201390 (2013)"},{"key":"7_CR31","doi-asserted-by":"crossref","unstructured":"Weninger, F., Hershey, J.R., Le\u00a0Roux, J., Schuller, B.: Discriminatively trained recurrent neural networks for single-channel speech separation. In: Proceedings of the IEEE Global Conference on Signal and Information Processing (GlobalSIP), pp.\u00a0577\u2013581 (2014)","DOI":"10.1109\/GlobalSIP.2014.7032183"},{"key":"7_CR32","volume-title":"Discriminative NMF and its application to single-channel source separation","author":"F. Weninger","year":"2014","unstructured":"Weninger, F., Le Roux, J., Hershey, J., Watanabe, S.: Discriminative NMF and its application to single-channel source separation. In: Proceedings of the Interspeech, Singapore (2014)"},{"key":"7_CR33","doi-asserted-by":"crossref","unstructured":"Weninger, F., Erdogan, H., Watanabe, S., Vincent, E., Le Roux, J., Hershey, J.R., Schuller, B.: Speech enhancement with LSTM recurrent neural networks and its application to noise-robust ASR. In: Proceedings of the International Conference on Latent Variable Analysis and Signal Separation (LVA\/ICA) (2015)","DOI":"10.1007\/978-3-319-22482-4_11"},{"issue":"3","key":"7_CR34","doi-asserted-by":"crossref","first-page":"483","DOI":"10.1109\/TASLP.2015.2512042","volume":"24","author":"D.S. Williamson","year":"2016","unstructured":"Williamson, D.S., Wang, Y., Wang, D.: Complex ratio masking for monaural speech separation. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(3), 483\u2013492 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"7_CR35","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y. Xu","year":"2014","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: An experimental study on speech enhancement based on deep neural networks. Signal Process. Lett. 21(1), 65\u201368 (2014)","journal-title":"Signal Process. Lett."}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T22:19:41Z","timestamp":1659737981000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_7","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}