{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T04:14:43Z","timestamp":1750997683713,"version":"3.41.0"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_4","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"79-104","source":"Crossref","is-referenced-by-count":0,"title":["Discriminative Beamforming with Phase-Aware Neural Networks for Speech Enhancement and Recognition"],"prefix":"10.1007","author":[{"given":"Xiong","family":"Xiao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hakan","family":"Erdogan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael","family":"Mandel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John R.","family":"Hershey","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael L.","family":"Seltzer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoguo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"4_CR1","unstructured":"Agarwal, A., Akchurin, E., Basoglu, C., Chen, G., Cyphers, S., Droppo, J., Eversole, A., Guenter, B., Hillebrand, M., Hoens, R., et\u00a0al.: An introduction to computational networks and the computational network toolkit. Microsoft Technical Report, MSR-TR-2014-112 (2014)"},{"issue":"4","key":"4_CR2","doi-asserted-by":"crossref","first-page":"943","DOI":"10.1121\/1.382599","volume":"65","author":"J. Allen","year":"1979","unstructured":"Allen, J., Berkley, D.: Image method for efficiently simulating small-room acoustics. J. Acoust. Soc. Am. 65(4), 943\u2013950 (1979)","journal-title":"J. Acoust. Soc. Am."},{"issue":"7","key":"4_CR3","doi-asserted-by":"crossref","first-page":"2011","DOI":"10.1109\/TASL.2007.902460","volume":"15","author":"X. Anguera","year":"2007","unstructured":"Anguera, X., Wooters, C., Hernando, J.: Acoustic beamforming for speaker diarization of meetings. IEEE Trans. Audio Speech Lang. Process. 15(7), 2011\u20132022 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"4_CR4","doi-asserted-by":"crossref","unstructured":"Barker, J., Marxer, R., Vincent, E., Watanabe, S.: The third \u201cCHiME\u201d speech separation and recognition challenge: dataset, task and baselines. In: 2015 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU 2015) (2015)","DOI":"10.1109\/ASRU.2015.7404837"},{"key":"4_CR5","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1007\/978-3-662-04619-7_2","volume-title":"Microphone Arrays: Signal Processing Techniques and Applications, Chap. 2","author":"J. Bitzer","year":"2001","unstructured":"Bitzer, J., Simmer, K.U.: Superdirective microphone arrays. In: Brandstein, M.S., Ward, D. (eds.) Microphone Arrays: Signal Processing Techniques and Applications, Chap.\u20092, pp. 19\u201338. Springer, Berlin (2001)"},{"issue":"8","key":"4_CR6","doi-asserted-by":"crossref","first-page":"1408","DOI":"10.1109\/PROC.1969.7278","volume":"57","author":"J. Capon","year":"1969","unstructured":"Capon, J.: High-resolution frequency-wavenumber spectrum analysis. Proc. IEEE 57(8), 1408\u20131418 (1969)","journal-title":"Proc. IEEE"},{"issue":"9","key":"4_CR7","doi-asserted-by":"crossref","first-page":"2230","DOI":"10.1109\/TSP.2002.801937","volume":"50","author":"S. Doclo","year":"2002","unstructured":"Doclo, S., Moonen, M.: GSVD-based optimal filtering for single and multimicrophone speech enhancement. IEEE Trans. Signal Process. 50(9), 2230\u20132244 (2002)","journal-title":"IEEE Trans. Signal Process."},{"key":"4_CR8","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1007\/978-3-662-04619-7_4","volume-title":"Microphone Arrays: Signal Processing Techniques and Applications, Chap. 4","author":"G.W. Elko","year":"2001","unstructured":"Elko, G.W.: Spatial coherence functions for differential microphones in isotropic noise fields. In: Brandstein, M.S., Ward, D. (eds.) Microphone Arrays: Signal Processing Techniques and Applications, Chap.\u20094, pp.\u00a061\u201385. Springer, Berlin (2001)"},{"issue":"6","key":"4_CR9","doi-asserted-by":"crossref","first-page":"1378","DOI":"10.1109\/TASSP.1983.1164219","volume":"31","author":"M. Er","year":"1983","unstructured":"Er, M., Cantoni, A.: Solar wind monitor satellite: derivative constraints for broad-band element space antenna array processors. IEEE Trans. Audio Speech Lang. Process. 31(6), 1378\u20131393 (1983)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"2","key":"4_CR10","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"M.J. Gales","year":"1998","unstructured":"Gales, M.J.: Maximum likelihood linear transformations for HMM-based speech recognition. Comput. Speech Lang. 12(2), 75\u201398 (1998)","journal-title":"Comput. Speech Lang."},{"issue":"1","key":"4_CR11","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1109\/TAP.1982.1142739","volume":"30","author":"L.J. Griffiths","year":"1982","unstructured":"Griffiths, L.J., Jim, C.W.: An alternative approach to linearly constrained adaptive beamforming. IEEE Trans. Antennas Propag. 30(1), 27\u201334 (1982)","journal-title":"IEEE Trans. Antennas Propag."},{"key":"4_CR12","unstructured":"Haeb-Umbach, R., Warsitz, E.: Adaptive filter-and-sum beamforming in spatially correlated noise. In: International Workshop on Acoustic Echo and Noise Control (IWAENC 2005) (2005)"},{"key":"4_CR13","doi-asserted-by":"crossref","unstructured":"Heymann, J., Drude, L., Chinaev, A., Haeb-Umbach, R.: BLSTM supported GEV beamformer front-end for the 3rd CHiME challenge. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a0444\u2013451. IEEE, New York (2015)","DOI":"10.1109\/ASRU.2015.7404829"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Hoshen, Y., Weiss, R.J., Wilson, K.W.: Speech acoustic modeling from raw multichannel waveforms. In: IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a04624\u20134628. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178847"},{"key":"4_CR15","volume-title":"Neural network based spectral mask estimation for acoustic beamforming","author":"L.D. Jahn Heymann","year":"2016","unstructured":"Jahn\u00a0Heymann, L.D., Haeb-Umbach, R.: Neural network based spectral mask estimation for acoustic beamforming. In: IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE, New York (2016)"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Kinoshita, K., Delcroix, M., Yoshioka, T., Nakatani, T., Sehr, A., Kellermann, W., Maas, R.: The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech. In: IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), pp.\u00a01\u20134. IEEE, New York (2013)","DOI":"10.1109\/WASPAA.2013.6701894"},{"issue":"4","key":"4_CR17","doi-asserted-by":"crossref","first-page":"320","DOI":"10.1109\/TASSP.1976.1162830","volume":"24","author":"C.H. Knapp","year":"1976","unstructured":"Knapp, C.H., Carter, G.C.: The generalized correlation method for estimation of time delay. IEEE Trans. Acoust. Speech Signal Process. 24(4), 320\u2013327 (1976)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"Liu, Y., Zhang, P., Hain, T.: Using neural network front-ends on far field multiple microphones based speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a05542\u20135546. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854663"},{"key":"4_CR19","doi-asserted-by":"crossref","unstructured":"Narayanan, A., Wang, D.: Joint noise adaptive training for robust automatic speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a02504\u20132508. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854051"},{"issue":"9","key":"4_CR20","doi-asserted-by":"crossref","first-page":"1215","DOI":"10.1109\/5.237532","volume":"81","author":"J.W. Picone","year":"1993","unstructured":"Picone, J.W.: Signal modeling techniques in speech recognition. Proc. IEEE 81(9), 1215\u20131247 (1993)","journal-title":"Proc. IEEE"},{"key":"4_CR21","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsky, J., Stemmer, G., Vesely, K.: The Kaldi speech recognition toolkit. In: IEEE 2011 Workshop on Automatic Speech Recognition and Understanding. IEEE Signal Processing Society (2011). IEEE Catalog No.: CFP11SRW-USB"},{"key":"4_CR22","doi-asserted-by":"crossref","unstructured":"Renals, S., Hain, T., Bourlard, H.: Recognition and understanding of meetings: the AMI and AMIDA projects. In: IEEE Workshop on Automatic Speech Recognition and Understanding, ASRU, Kyoto (2007). IDIAP-RR 07-46","DOI":"10.1109\/ASRU.2007.4430116"},{"key":"4_CR23","doi-asserted-by":"crossref","unstructured":"Robinson, T., Fransen, J., Pye, D., Foote, J., Renals, S.: WSJCAM0: a British English speech corpus for large vocabulary continuous speech recognition. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a081\u201384 (1995)","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"4_CR24","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Weiss, R.J., Wilson, K.W., Narayanan, A., Bacchiani, M., Senior, A.: Speaker location and microphone spacing invariant acoustic modeling from raw multichannel waveforms. In: IEEE Workshop on Automatic Speech Recognition and Understanding (ARSU), pp.\u00a030\u201336 (2015)","DOI":"10.1109\/ASRU.2015.7404770"},{"key":"4_CR25","volume-title":"Factored spatial and spectral multichannel raw waveform CLDNNs","author":"T.N. Sainath","year":"2016","unstructured":"Sainath, T.N., Weiss, R.J., Wilson, K.W., Narayanan, A., Bacchiani, M.: Factored spatial and spectral multichannel raw waveform CLDNNs. In: IEEE International Conference on Acoustics, Speech and Signal Processing (2016)"},{"issue":"5","key":"4_CR26","doi-asserted-by":"crossref","first-page":"489","DOI":"10.1109\/TSA.2004.832988","volume":"12","author":"M.L. Seltzer","year":"2004","unstructured":"Seltzer, M.L., Raj, B., Stern, R.M.: Likelihood-maximizing beamforming for robust hands-free speech recognition. IEEE Trans. Speech Audio Process. 12(5), 489\u2013498 (2004)","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"2","key":"4_CR27","doi-asserted-by":"crossref","first-page":"260","DOI":"10.1109\/TASL.2009.2025790","volume":"18","author":"M. Souden","year":"2010","unstructured":"Souden, M., Benesty, J., Affes, S.: On optimal frequency-domain multichannel linear filtering for noise reduction. IEEE Trans. Audio Speech Lang. Process. 18(2), 260\u2013276 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"9","key":"4_CR28","doi-asserted-by":"crossref","first-page":"1120","DOI":"10.1109\/LSP.2014.2325781","volume":"21","author":"P. Swietojanski","year":"2014","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Convolutional neural networks for distant speech recognition. IEEE Signal Process Lett. 21(9), 1120\u20131124 (2014)","journal-title":"IEEE Signal Process Lett."},{"issue":"2","key":"4_CR29","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1109\/53.665","volume":"5","author":"B.D. Veen Van","year":"1988","unstructured":"Van\u00a0Veen, B.D., Buckley, K.M.: Beamforming: a versatile approach to spatial filtering. IEEE ASSP Mag. 5(2), 4\u201324 (1988)","journal-title":"IEEE ASSP Mag."},{"key":"4_CR30","doi-asserted-by":"crossref","unstructured":"Xiao, X., Zhao, S., Zhong, X., Jones, D.L., Chng, E.S., Li, H.: A learning-based approach to direction of arrival estimation in noisy and reverberant environments. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a02814\u20132818. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178484"},{"key":"4_CR31","unstructured":"Xiao, X., Xu, C., Zhang, Z., Zhao, S., Sun, S., Watanabe, S., Wang, L., Xie, L., Jones, D.L., Chng, E.S., Li, H.: Investigation of neural networks based beamforming approaches for speech recognition: the NTU systems for CHiME-4 evaluation. In: CHiME 4 Workshop (2016)"},{"key":"4_CR32","unstructured":"Young, S., Evermann, G., Gales, M., Hain, T., Kershaw, D., Liu, X., Moore, G., Odell, J., Ollason, D., Povey, D., et\u00a0al.: The HTK Book, 3.4 edn. Cambridge University Engineering Department, Cambridge (2006)"},{"key":"4_CR33","volume-title":"An introduction to computational networks and the computational network toolkit","author":"D. Yu","year":"2014","unstructured":"Yu, D., Eversole, A., Seltzer, M., Yao, K., Huang, Z., Guenter, B., Kuchaiev, O., Zhang, Y., Seide, F., Wang, H., et\u00a0al.: An introduction to computational networks and the computational network toolkit. Tech. Rep. MSR, Microsoft Research (2014)"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T20:15:55Z","timestamp":1750968955000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_4","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}