{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T00:49:16Z","timestamp":1740098956140,"version":"3.37.3"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_2","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T04:37:26Z","timestamp":1509424646000},"page":"21-49","source":"Crossref","is-referenced-by-count":3,"title":["Multichannel Speech Enhancement Approaches to DNN-Based Far-Field Speech Recognition"],"prefix":"10.1007","author":[{"given":"Marc","family":"Delcroix","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takuya","family":"Yoshioka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nobutaka","family":"Ito","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Atsunori","family":"Ogawa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keisuke","family":"Kinoshita","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masakiyo","family":"Fujimoto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takuya","family":"Higuchi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shoko","family":"Araki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomohiro","family":"Nakatani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"unstructured":"Anguera, X.: BeamformIt. http:\/\/www.xavieranguera.com\/beamformit\/ (2014)","key":"2_CR1"},{"issue":"7","key":"2_CR2","doi-asserted-by":"crossref","first-page":"2011","DOI":"10.1109\/TASL.2007.902460","volume":"15","author":"X. Anguera","year":"2007","unstructured":"Anguera, X., Wooters, C., Hernando, J.: Acoustic beamforming for speaker diarization of meetings. IEEE Trans. Audio Speech Lang. Process. 15(7), 2011\u20132023 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Araki, S., Sawada, H., Makino, S.: Blind speech separation in a meeting situation with maximum SNR beamformers. In: Proceedings of ICASSP\u201907, vol.\u00a01, pp.\u00a0I-41\u2013I-44 (2007)","key":"2_CR3","DOI":"10.1109\/ICASSP.2007.366611"},{"doi-asserted-by":"crossref","unstructured":"Araki, S., Okada, M., Higuchi, T., Ogawa, A., Nakatani, T.: Spatial correlation model based observation vector clustering and MVDR beamforming for meeting recognition. In: Proceedings of ICASSP\u201916, pp.\u00a0385\u2013389 (2016)","key":"2_CR4","DOI":"10.1109\/ICASSP.2016.7471702"},{"doi-asserted-by":"crossref","unstructured":"Barker, J., Marxer, R., Vincent, E., Watanabe, S.: The third \u201cCHiME\u201d speech separation and recognition challenge: dataset, task and baselines. In: Proceedings of ASRU\u201915, pp.\u00a0504\u2013511 (2015)","key":"2_CR5","DOI":"10.1109\/ASRU.2015.7404837"},{"key":"2_CR6","volume-title":"Pattern Recognition and Machine Learning","author":"C.M. Bishop","year":"2006","unstructured":"Bishop, C.M.: Pattern Recognition and Machine Learning. Information Science and Statistics. Springer, New York (2006)"},{"issue":"2","key":"2_CR7","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1109\/TASSP.1979.1163209","volume":"27","author":"S. Boll","year":"1979","unstructured":"Boll, S.: Suppression of acoustic noise in speech using spectral subtraction. IEEE Trans. Acoust. Speech Signal Process. 27(2), 113\u2013120 (1979)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"6","key":"2_CR8","doi-asserted-by":"crossref","first-page":"3233","DOI":"10.1121\/1.1570439","volume":"113","author":"J.S. Bradley","year":"2003","unstructured":"Bradley, J.S., Sato, H., Picard, M.: On the importance of early reflections for speech in rooms. J. Acoust. Soc. Am. 113(6), 3233\u20133244 (2003)","journal-title":"J. Acoust. Soc. Am."},{"key":"2_CR9","first-page":"69","volume":"2008","author":"A. Brutti","year":"2008","unstructured":"Brutti, A., Omologo, M., Svaizer, P.: Comparison between different sound source localization techniques based on a real data collection. In: Hands-Free Speech Communication and Microphone Arrays, 2008, HSCMA 2008, pp.\u00a069\u201372 (2008)","journal-title":"HSCMA"},{"key":"2_CR10","volume-title":"The AMI Meeting Corpus: A Pre-announcement","author":"J. Carletta","year":"2005","unstructured":"Carletta, J., Ashby, S., Bourban, S., Flynn, M., Guillemot, M., Hain, T., Kadlec, J., Karaiskos, V., Kraaij, W., Kronenthal, M., et\u00a0al.: The AMI Meeting Corpus: A Pre-announcement. Springer, Berlin (2005)"},{"key":"2_CR11","doi-asserted-by":"publisher","first-page":"170","DOI":"10.1155\/ASP\/2006\/26503","volume":"2006","author":"J. Chen","year":"2006","unstructured":"Chen, J., Benesty, J., Huang, Y.: Time delay estimation in room acoustic environments: an overview. EURASIP J. Adv. Signal Process. 2006, 170\u2013170 (2006). doi:10.1155\/ASP\/2006\/26503. http:\/\/dx.doi.org\/10.1155\/ASP\/2006\/26503","journal-title":"EURASIP J. Adv. Signal Process."},{"doi-asserted-by":"crossref","unstructured":"Delcroix, M., Yoshioka, T., Ogawa, A., Kubo, Y., Fujimoto, M., Ito, N., Kinoshita, K., Espi, M., Araki, S., Hori, T., Nakatani, T.: Strategies for distant speech recognition in reverberant environments. EURASIP J. Adv. Signal Process. 2015, 60 (2015). doi:10.1186\/s13634-015-0245-7","key":"2_CR12","DOI":"10.1186\/s13634-015-0245-7"},{"doi-asserted-by":"crossref","unstructured":"Dennis, J., Dat, T.H.: Single and multi-channel approaches for distant speech recognition under noisy reverberant conditions: I2R\u2019S system description for the ASpIRE challenge. In: Proceedings of ASRU\u201915, pp.\u00a0518\u2013524 (2015)","key":"2_CR13","DOI":"10.1109\/ASRU.2015.7404839"},{"issue":"9","key":"2_CR14","doi-asserted-by":"crossref","first-page":"2230","DOI":"10.1109\/TSP.2002.801937","volume":"50","author":"S. Doclo","year":"2002","unstructured":"Doclo, S., Moonen, M.: GSVD-based optimal filtering for single and multimicrophone speech enhancement. IEEE Trans. Signal Process. 50(9), 2230\u20132244 (2002)","journal-title":"IEEE Trans. Signal Process."},{"doi-asserted-by":"crossref","unstructured":"Erdogan, H., Hershey, J.R., Watanabe, S., Le\u00a0Roux, J.: Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks. In: Proceedings of ICASSP\u201915, pp.\u00a0708\u2013712 (2015)","key":"2_CR15","DOI":"10.1109\/ICASSP.2015.7178061"},{"issue":"8","key":"2_CR16","doi-asserted-by":"crossref","first-page":"926","DOI":"10.1109\/PROC.1972.8817","volume":"60","author":"O.L. Frost","year":"1972","unstructured":"Frost, O.L.: An algorithm for linearly constrained adaptive array processing. Proc. IEEE 60(8), 926\u2013935 (1972)","journal-title":"Proc. IEEE"},{"doi-asserted-by":"crossref","unstructured":"Harper, M.: The automatic speech recognition in reverberant environments (ASpIRE) challenge. In: Proceedings of ASRU\u201915, pp.\u00a0547\u2013554 (2015)","key":"2_CR17","DOI":"10.1109\/ASRU.2015.7404843"},{"key":"2_CR18","volume-title":"Adaptive Filter Theory","author":"S. Haykin","year":"1996","unstructured":"Haykin, S.: Adaptive Filter Theory, 3rd edn. Prentice-Hall, Upper Saddle River, NJ (1996)","edition":"3"},{"doi-asserted-by":"crossref","unstructured":"Heymann, J., Drude, L., Chinaev, A., Haeb-Umbach, R.: BLSTM supported GEV beamformer front-end for the 3RD CHiME challenge. In: Proceedings of ASRU\u201915, pp.\u00a0444\u2013451. IEEE, New York (2015)","key":"2_CR19","DOI":"10.1109\/ASRU.2015.7404829"},{"doi-asserted-by":"crossref","unstructured":"Higuchi, T., Ito, N., Yoshioka, T., Nakatani, T.: Robust MVDR beamforming using time-frequency masks for online\/offline ASR in noise. In: Proceedings of ICASSP\u201916, pp.\u00a05210\u20135214 (2016)","key":"2_CR20","DOI":"10.1109\/ICASSP.2016.7472671"},{"issue":"2","key":"2_CR21","doi-asserted-by":"crossref","first-page":"499","DOI":"10.1109\/TASL.2011.2164527","volume":"20","author":"T. Hori","year":"2012","unstructured":"Hori, T., Araki, S., Yoshioka, T., Fujimoto, M., Watanabe, S., Oba, T., Ogawa, A., Otsuka, K., Mikami, D., Kinoshita, K., Nakatani, T., Nakamura, A., Yamato, J.: Low-latency real-time meeting recognition and understanding using distant microphones and omni-directional camera. IEEE Trans. Audio Speech Lang. Process. 20(2), 499\u2013513 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Hori, T., Chen, Z., Erdogan, H., Hershey, J.R., Roux, J., Mitra, V., Watanabe, S.: The MERL\/SRI system for the 3rd CHiME challenge using beamforming, robust feature extraction, and advanced speech recognition. In: Proceedings of ASRU\u201915, pp.\u00a0475\u2013481 (2015)","key":"2_CR22","DOI":"10.1109\/ASRU.2015.7404833"},{"key":"2_CR23","volume-title":"Spoken Language Processing: A Guide to Theory, Algorithm, and System Development","author":"X. Huang","year":"2001","unstructured":"Huang, X., Acero, A., Hon, H.W.: Spoken Language Processing: A Guide to Theory, Algorithm, and System Development, 1st edn. Prentice-Hall, Upper Saddle River, NJ (2001)","edition":"1"},{"doi-asserted-by":"crossref","unstructured":"Jukic, A., Doclo, S.: Speech dereverberation using weighted prediction error with Laplacian model of the desired signal. In: Proceedings of ICASSP\u201914, pp.\u00a05172\u20135176 (2014)","key":"2_CR24","DOI":"10.1109\/ICASSP.2014.6854589"},{"issue":"4","key":"2_CR25","doi-asserted-by":"crossref","first-page":"534","DOI":"10.1109\/TASL.2008.2009015","volume":"17","author":"K. Kinoshita","year":"2009","unstructured":"Kinoshita, K., Delcroix, M., Nakatani, T., Miyoshi, M.: Suppression of late reverberation effect on speech signal using long-term multiple-step linear prediction. IEEE Trans. Audio Speech Lang. Process. 17(4), 534\u2013545 (2009)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Kinoshita, K., Delcroix, M., Yoshioka, T., Nakatani, T., Habets, E., Sehr, A., Kellermann, W., Gannot, S., Maas, R., Haeb-Umbach, R., Leutnant, V., Raj, B.: The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech. In: Proceedings of WASPAA\u201913. New Paltz, NY (2013)","key":"2_CR26","DOI":"10.1109\/WASPAA.2013.6701894"},{"key":"2_CR27","doi-asserted-by":"publisher","DOI":"10.1186\/s13634-016-0306-6","author":"K. Kinoshita","year":"2016","unstructured":"Kinoshita, K., Delcroix, M., Gannot, S., Habets, E., Haeb-Umbach, R., Kellermann, W., Leutnant, V., Maas, R., Nakatani, T., Raj, B., Sehr, A., Yoshioka, T.: A summary of the REVERB challenge: state-of-the-art and remaining challenges in reverberant speech processing research. EURASIP J. Adv. Signal Process. (2016). doi:10.1186\/s13634-016-0306-6","journal-title":"EURASIP J. Adv. Signal Process."},{"key":"2_CR28","volume-title":"Room Acoustics","author":"H. Kuttruff","year":"2009","unstructured":"Kuttruff, H.: Room Acoustics, 5th edn. Taylor & Francis, London (2009)","edition":"5"},{"issue":"3","key":"2_CR29","first-page":"359","volume":"87","author":"K. Lebart","year":"2001","unstructured":"Lebart, K., Boucher, J.M., Denbigh, P.N.: A new method based on spectral subtraction for speech dereverberation. Acta Acustica 87(3), 359\u2013366 (2001)","journal-title":"Acta Acustica"},{"doi-asserted-by":"crossref","unstructured":"Nakatani, T., Yoshioka, T., Kinoshita, K., Miyoshi, M., Juang, B.H.: Blind speech dereverberation with multi-channel linear prediction based on short time Fourier transform representation. In: Proceedings of ICASSP\u201908, pp.\u00a085\u201388 (2008)","key":"2_CR30","DOI":"10.1109\/ICASSP.2008.4517552"},{"issue":"7","key":"2_CR31","doi-asserted-by":"crossref","first-page":"1717","DOI":"10.1109\/TASL.2010.2052251","volume":"18","author":"T. Nakatani","year":"2010","unstructured":"Nakatani, T., Yoshioka, T., Kinoshita, K., Miyoshi, M., Juang, B.H.: Speech dereverberation based on variance-normalized delayed linear prediction. IEEE Trans. Audio Speech Lang. Process. 18(7), 1717\u20131731 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Narayanan, A., Wang, D.: Ideal ratio mask estimation using deep neural networks for robust speech recognition. In: Proceedings of ICASSP\u201913, pp.\u00a07092\u20137096. IEEE, New York (2013)","key":"2_CR32","DOI":"10.1109\/ICASSP.2013.6639038"},{"doi-asserted-by":"crossref","unstructured":"Renals, S., Swietojanski, P.: Neural networks for distant speech recognition. In: 2014 4th Joint Workshop on Hands-free Speech Communication and Microphone Arrays (HSCMA), pp.\u00a0172\u2013176 (2014)","key":"2_CR33","DOI":"10.1109\/HSCMA.2014.6843274"},{"doi-asserted-by":"crossref","unstructured":"Sivasankaran, S., Nugraha, A.A., Vincent, E., Morales-Cordovilla, J.A., Dalmia, S., Illina, I., Liutkus, A.: Robust ASR using neural network based speech enhancement and feature simulation. In: Proceedings of ASRU\u201915, pp.\u00a0482\u2013489 (2015)","key":"2_CR34","DOI":"10.1109\/ASRU.2015.7404834"},{"issue":"9","key":"2_CR35","doi-asserted-by":"crossref","first-page":"1913","DOI":"10.1109\/TASL.2013.2263137","volume":"21","author":"M. Souden","year":"2013","unstructured":"Souden, M., Araki, S., Kinoshita, K., Nakatani, T., Sawada, H.: A multichannel MMSE-based framework for speech source separation and noise reduction. IEEE Trans. Audio Speech Lang. Process. 21(9), 1913\u20131928 (2013)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"unstructured":"Tachioka, Y., Narita, T., Weninger, F., Watanabe, S.: Dual system combination approach for various reverberant environments with dereverberation techniques. In: Proceedings of REVERB\u201914 (2014)","key":"2_CR36"},{"key":"2_CR37","volume-title":"Detection, Estimation, and Modulation Theory","author":"H.L. Trees Van","year":"2002","unstructured":"Van\u00a0Trees, H.L.: Detection, Estimation, and Modulation Theory. Part IV, Optimum Array Processing. Wiley-Interscience, New York (2002)"},{"issue":"3","key":"2_CR38","doi-asserted-by":"crossref","first-page":"1066","DOI":"10.1109\/TASL.2006.885253","volume":"15","author":"T. Virtanen","year":"2007","unstructured":"Virtanen, T.: Monaural sound source separation by nonnegative matrix factorization with temporal continuity and sparseness criteria. IEEE Trans. Audio Speech Lang. Process. 15(3), 1066\u20131074 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"5","key":"2_CR39","doi-asserted-by":"crossref","first-page":"1529","DOI":"10.1109\/TASL.2007.898454","volume":"15","author":"E. Warsitz","year":"2007","unstructured":"Warsitz, E., Haeb-Umbach, R.: Blind acoustic beamforming based on generalized eigenvalue decomposition. IEEE Trans. Audio Speech Lang. Process. 15(5), 1529\u20131539 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"unstructured":"Weninger, F., Watanabe, S., Roux, J.L., Hershey, J.R., Tachioka, Y., Geiger, J., Schuller, B., Rigoll, G.: The MERL\/MELCO\/TUM system for the REVERB challenge using deep recurrent neural network feature enhancement. In: Proceedings of REVERB\u201914 (2014)","key":"2_CR40"},{"doi-asserted-by":"crossref","unstructured":"Weninger, F., Erdogan, H., Watanabe, S., Vincent, E., Le\u00a0Roux, J., Hershey, J.R., Schuller, B.: Speech enhancement with LSTM recurrent neural networks and its application to noise-robust ASR. In: Proceedings of Latent Variable Analysis and Signal Separation, pp.\u00a091\u201399. Springer, Berlin (2015)","key":"2_CR41","DOI":"10.1007\/978-3-319-22482-4_11"},{"issue":"1","key":"2_CR42","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y. Xu","year":"2015","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(1), 7\u201319 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"10","key":"2_CR43","doi-asserted-by":"crossref","first-page":"2707","DOI":"10.1109\/TASL.2012.2210879","volume":"20","author":"T. Yoshioka","year":"2012","unstructured":"Yoshioka, T., Nakatani, T.: Generalization of multi-channel linear prediction methods for blind MIMO impulse response shortening. IEEE Trans. Audio Speech Lang. Process. 20(10), 2707\u20132720 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Tachibana, H., Nakatani, T., Miyoshi, M.: Adaptive dereverberation of speech signals with speaker-position change detection. In: 2009 IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a03733\u20133736 (2009)","key":"2_CR44","DOI":"10.1109\/ICASSP.2009.4960438"},{"issue":"2","key":"2_CR45","doi-asserted-by":"crossref","first-page":"231","DOI":"10.1109\/TASL.2008.2008042","volume":"17","author":"T. Yoshioka","year":"2009","unstructured":"Yoshioka, T., Nakatani, T., Miyoshi, M.: Integrated speech enhancement method using noise suppression and dereverberation. IEEE Trans. Audio Speech Lang. Process. 17(2), 231\u2013246 (2009)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"6","key":"2_CR46","doi-asserted-by":"crossref","first-page":"114","DOI":"10.1109\/MSP.2012.2205029","volume":"29","author":"T. Yoshioka","year":"2012","unstructured":"Yoshioka, T., Sehr, A., Delcroix, M., Kinoshita, K., Maas, R., Nakatani, T., Kellermann, W.: Making machines understand us in reverberant rooms: robustness against reverberation for automatic speech recognition. IEEE Signal Process. Mag. 29(6), 114\u2013126 (2012)","journal-title":"IEEE Signal Process. Mag."},{"doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Chen, X., Gales, M.J.F.: Impact of single-microphone dereverberation on DNN-based meeting transcription systems. In: Proceedings of ICASSP\u201914 (2014)","key":"2_CR47","DOI":"10.1109\/ICASSP.2014.6854660"},{"doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Ito, N., Delcroix, M., Ogawa, A., Kinoshita, K., Fujimoto, M., Yu, C., Fabian, W.J., Espi, M., Higuchi, T., Araki, S., Nakatani, T.: The NTT CHiME-3 system: advances in speech enhancement and recognition for mobile multi-microphone devices. In: Proceedings of ASRU\u201915, pp.\u00a0436\u2013443 (2015)","key":"2_CR48","DOI":"10.1109\/ASRU.2015.7404828"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,10,5]],"date-time":"2019-10-05T06:42:38Z","timestamp":1570257758000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_2","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}