{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T10:22:48Z","timestamp":1771064568552,"version":"3.50.1"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2016,1,13]],"date-time":"2016-01-13T00:00:00Z","timestamp":1452643200000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["EURASIP J. Adv. Signal Process."],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1186\/s13634-015-0300-4","type":"journal-article","created":{"date-parts":[[2016,1,13]],"date-time":"2016-01-13T16:28:18Z","timestamp":1452702498000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":24,"title":["Speech dereverberation for enhancement and recognition using dynamic features constrained deep neural networks and feature adaptation"],"prefix":"10.1186","volume":"2016","author":[{"given":"Xiong","family":"Xiao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengkui","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Duc Hoang","family":"Ha Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xionghu","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Douglas L.","family":"Jones","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eng Siong","family":"Chng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haizhou","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,1,13]]},"reference":[{"issue":"6","key":"300_CR1","doi-asserted-by":"publisher","first-page":"575","DOI":"10.1111\/j.1467-9892.1993.tb00167.x","volume":"14","author":"TH Li","year":"1993","unstructured":"TH Li, Estimation and blind deconvolution of autoregressive systems with nonstationary binary inputs. J. Time Ser. Anal.14(6), 575\u2013588 (1993).","journal-title":"J. Time Ser. Anal."},{"key":"300_CR2","doi-asserted-by":"publisher","first-page":"2410","DOI":"10.1109\/78.469847","volume":"43","author":"R Chen","year":"1995","unstructured":"R Chen, TH Li, Blind restoration of linearly degraded discrete signals by gibbs sampling. IEEE Trans. Signal Process.43:, 2410\u20132413 (1995).","journal-title":"IEEE Trans. Signal Process."},{"issue":"1","key":"300_CR3","first-page":"3","volume":"73","author":"O Cappe","year":"1999","unstructured":"O Cappe, A Doucet, M Lavielle, E Moulines, Simulation-based methods for blind maximum-likelihood filter deconvolution. IEEE Trans. Signal Process.73(1), 3\u201325 (1999).","journal-title":"IEEE Trans. Signal Process."},{"issue":"11","key":"300_CR4","doi-asserted-by":"publisher","first-page":"1074","DOI":"10.1155\/S1110865703305049","volume":"2003","author":"S Gannot","year":"2003","unstructured":"S Gannot, M Moonen, Subspace methods for multimicrophone speech dereverberation. EURASIP J. Appl. Signal Process.2003(11), 1074\u20131090 (2003).","journal-title":"EURASIP J. Appl. Signal Process."},{"key":"300_CR5","doi-asserted-by":"crossref","unstructured":"M Triki, DTM Slock, in Proc. IEEE International Conference on Acoustics, Speech, and Signal Processing, 5. Delay and predict equalization for blind speech dereverberation (Toulouse, France, 2006), pp. 97\u2013100.","DOI":"10.1109\/ICASSP.2006.1661221"},{"issue":"2","key":"300_CR6","doi-asserted-by":"publisher","first-page":"430","DOI":"10.1109\/TASL.2006.881698","volume":"15","author":"M Delcroix","year":"2006","unstructured":"M Delcroix, T Hikichi, M Miyoshi, Precise dereverberation using multichannel linear prediction. IEEE Trans. Audio, Speech, Lang. Process.15(2), 430\u2013440 (2006).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"issue":"5","key":"300_CR7","doi-asserted-by":"publisher","first-page":"392","DOI":"10.1109\/89.536934","volume":"4","author":"S Subramaniam","year":"1996","unstructured":"S Subramaniam, A Petropulu, C Wendt, Cepstrum-based deconvolution for speech dereverberation. IEEE Trans. Speech Audio Process.4(5), 392\u2013396 (1996).","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"2","key":"300_CR8","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/53.665","volume":"5","author":"BDV Veen","year":"1988","unstructured":"BDV Veen, KM Buckley, Beamforming: A versatile approach to spatial filtering. IEEE ASSP Mag. 5(2), 4\u201324 (1988).","journal-title":"IEEE ASSP Mag"},{"key":"300_CR9","doi-asserted-by":"publisher","first-page":"912","DOI":"10.1121\/1.381621","volume":"62","author":"J Allen","year":"1977","unstructured":"J Allen, D Berkley, Multimicrophone signal processing technique to remove room reverberation from speech signals. J. Acoust. Soc. Am.62:, 912\u2013915 (1977).","journal-title":"J. Acoust. Soc. Am."},{"key":"300_CR10","doi-asserted-by":"crossref","unstructured":"R Zelinski, in Int. Conf. on Acoust. Speech and Sig. Proc. A microphone array with adaptive post-filtering for noise reduction in reverberant rooms (New York, USA, 1988), pp. 2578\u20132581.","DOI":"10.1109\/ICASSP.1988.197172"},{"key":"300_CR11","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/S0167-6393(96)00054-4","volume":"20","author":"S Fischer","year":"1996","unstructured":"S Fischer, Beamforming microphone arrays for speech acquisition in noisy environments. Speech Commun.20:, 215\u2013227 (1996).","journal-title":"Speech Commun."},{"issue":"1","key":"300_CR12","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1109\/TASL.2009.2024731","volume":"18","author":"E Habets","year":"2010","unstructured":"E Habets, J Benesty, I Cohen, S Gannot, J Dmochowski, New insights into MVDR beamformer in room acoustics. IEEE Trans. Audio, Speech Lang. Process.18(1), 158\u2013170 (2010).","journal-title":"IEEE Trans. Audio, Speech Lang. Process."},{"issue":"5","key":"300_CR13","doi-asserted-by":"publisher","first-page":"945","DOI":"10.1109\/TASL.2013.2239292","volume":"21","author":"E Habets","year":"2013","unstructured":"E Habets, J Benesty, A two stage beamforming approach for noise reduction and dereverberation. IEEE Trans. Audio, Speech Lang. Process.21(5), 945\u2013958 (2013).","journal-title":"IEEE Trans. Audio, Speech Lang. Process."},{"issue":"3","key":"300_CR14","first-page":"359","volume":"87","author":"K Lebart","year":"2001","unstructured":"K Lebart, JM Boucher, PN Denbigh, A new method based on spectral subtraction for speech dereverberation. ACUSTICA. 87(3), 359\u2013366 (2001).","journal-title":"ACUSTICA"},{"key":"300_CR15","doi-asserted-by":"crossref","unstructured":"FS Pacheco, R Seara, in Proc. of the Fifth International Telecommunications Symposium (ITS2006), 4. Spectral subtraction for reverberation reduction applied to automatic speech recognition (Fortaleza-CE, Brazil, 2006), pp. 581\u2013584.","DOI":"10.1109\/ITS.2006.4433380"},{"issue":"1","key":"300_CR16","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1016\/j.csl.2014.11.008","volume":"31","author":"T Yoshioka","year":"2015","unstructured":"T Yoshioka, MJ Gales, Environmentally robust asr front-end for deep neural network acoustic models. Comput. Speech Lang.31(1), 65\u201386 (2015).","journal-title":"Comput. Speech Lang."},{"key":"300_CR17","doi-asserted-by":"crossref","unstructured":"L Deng, A Acero, M Plumpe, XD Huang, in Proc. ICSLP \u201900. Large-vocabulary speech recognition under adverse acoustic environment (Beijing, China, 2000), pp. 806\u2013809.","DOI":"10.21437\/ICSLP.2000-657"},{"key":"300_CR18","doi-asserted-by":"crossref","unstructured":"X Xiao, J Li, ES Chng, H Li, in IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). Feature compensation using linear combination of speaker and environment dependent correction vectors (Florence, Italy, 2014), pp. 1720\u20131724.","DOI":"10.1109\/ICASSP.2014.6853892"},{"issue":"8","key":"300_CR19","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TASL.2007.907344","volume":"15","author":"T Toda","year":"2007","unstructured":"T Toda, AW Black, K Tokuda, Voice conversion based on maximum-likelihood estimation of spectral parameters trajectory. IEEE Trans. Audio, Speech, Lang. Process.15(8), 2222\u20132235 (2007).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"key":"300_CR20","unstructured":"EA Wan, AT Nelson, in Handbook of neural networks for speech processing, ed. by S Katagiri. Networks for speech enhancement (Artech House, Boston, 1998)."},{"issue":"7","key":"300_CR21","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"8","author":"GE Hinton","year":"2006","unstructured":"GE Hinton, S Osindero, Y Teh, A fast learning algorithm for deep belief nets. Neural Comput. 8(7), 1527\u20131554 (2006).","journal-title":"Neural Comput"},{"key":"300_CR22","unstructured":"Y Bengio, 2. Foundations and Trends\u00ae; in Machine Learning. Learning deep architectures for AI, (2009), pp. 1\u2013127."},{"issue":"6","key":"300_CR23","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"GE Hinton","year":"2012","unstructured":"GE Hinton, L Deng, D Yu, GE Dahl, A Mohamed, N Jaitly, A Senior, V Vanhoucke, P Nguyen, T Sainath, B Kingsbury, Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. Signal Process. Mag. IEEE. 29(6), 82\u201397 (2012).","journal-title":"Signal Process. Mag. IEEE"},{"key":"300_CR24","volume-title":"Interspeech 2012","author":"AL Maas","year":"2012","unstructured":"AL Maas, QV Le, TM O\u2019Neil, O Vinyals, P Nguyen, AY Ng, in Interspeech 2012. Recurrent neural networks for noise reduction in robust asr (CiteseerPortland, Oregon, 2012)."},{"issue":"4","key":"300_CR25","doi-asserted-by":"publisher","first-page":"888","DOI":"10.1016\/j.csl.2014.01.001","volume":"28","author":"F Weninger","year":"2014","unstructured":"F Weninger, J Geiger, M W\u00f6llmer, B Schuller, G Rigoll, Feature enhancement by deep lstm networks for asr in reverberant multisource environments. Comput. Speech Lang.28(4), 888\u2013902 (2014).","journal-title":"Comput. Speech Lang."},{"issue":"8","key":"300_CR26","doi-asserted-by":"publisher","first-page":"1296","DOI":"10.1109\/TASLP.2014.2329237","volume":"22","author":"B Li","year":"2014","unstructured":"B Li, KC Sim, A spectral masking approach to noise-robust speech recognition using deep neural networks. IEEE\/ACM Trans. Audio, Speech Lang. Process. (TASLP). 22(8), 1296\u20131305 (2014).","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Process. (TASLP)"},{"issue":"1","key":"300_CR27","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y Xu","year":"2015","unstructured":"Y Xu, J Du, L-R Dai, C-H Lee, A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 23(1), 7\u201319 (2015).","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process"},{"key":"300_CR28","doi-asserted-by":"crossref","unstructured":"J Du, Q Wang, T Gao, Y Xu, L Dai, C-H Lee, in Interspeech 2014. Robust speech recognition with speech enhanced deep neural networks (Singapore, 2014).","DOI":"10.21437\/Interspeech.2014-148"},{"key":"300_CR29","unstructured":"X Xiao, S Zhao, DHH Nguyen, X Zhong, DL Jones, ES Chng, H Li, in Proceeding of REVERB challenge workshop. The NTU-ADSC systems for reverberation challenge (Florence, Italy, 2014)."},{"key":"300_CR30","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"MJF Gales","year":"1998","unstructured":"MJF Gales, Maximum likelihood linear transformations for HMM-based speech recognition. Comput. Speech Lang.12:, 75\u201398 (1998).","journal-title":"Comput. Speech Lang."},{"key":"300_CR31","doi-asserted-by":"crossref","unstructured":"DHH Nguyen, X Xiao, ES Chng, H Li, in ICASSP 2014. Generalization of temporal filter and linear transformation for robust speech recognition (Florence, Italy, 2014).","DOI":"10.1109\/ICASSP.2014.6853894"},{"key":"300_CR32","volume-title":"Room acoustics","author":"H Kuttruff","year":"2000","unstructured":"H Kuttruff, Room acoustics, 4th edn. (Taylor & Francis, 270 Madison Avenue, New York, NY, 2000)."},{"issue":"4","key":"300_CR33","doi-asserted-by":"publisher","first-page":"320","DOI":"10.1109\/TASSP.1976.1162830","volume":"24","author":"CH Knapp","year":"1976","unstructured":"CH Knapp, GC Carter, The generalized correlation method for estimation of time delay. IEEE Trans. Acoust., Speech Signal Process.24(4), 320\u2013327 (1976).","journal-title":"IEEE Trans. Acoust., Speech Signal Process."},{"issue":"8","key":"300_CR34","doi-asserted-by":"publisher","first-page":"926","DOI":"10.1109\/PROC.1972.8817","volume":"60","author":"OLF III","year":"1972","unstructured":"OLF III, An algorithm for linearly constrained adaptive array process. IEEE Proc.60(8), 926\u2013935 (1972).","journal-title":"IEEE Proc."},{"key":"300_CR35","unstructured":"HW L\u00f6llmann, E Yilmaz, M Jeub, P Vary, in International Workshop on Acoustic Echo and Noise Control (IWAENC). An improved algorithm for blind reverberation time estimation (Tel Aviv, Israel, 2010)."},{"issue":"1","key":"300_CR36","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1109\/TASSP.1986.1164788","volume":"34","author":"S Furui","year":"1986","unstructured":"S Furui, Speaker independent isolated word recognizer using dynamic features of speech spectrum. IEEE Trans. Acoustics, Speech Signal Process.34(1), 52\u201359 (1986).","journal-title":"IEEE Trans. Acoustics, Speech Signal Process."},{"issue":"2","key":"300_CR37","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1109\/89.279278","volume":"2","author":"JL Gauvain","year":"1994","unstructured":"JL Gauvain, CH Lee, Maximum a posteriori estimation for multivariate Gaussian mixture observations of Markov chains. IEEE Trans. Speech Audio Process.2(2), 291\u2013298 (1994).","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"300_CR38","unstructured":"PJ Moreno, Speech recognition in noisy environments. PhD thesis (ECE, Carnegie Mellon University, 1996)."},{"key":"300_CR39","doi-asserted-by":"crossref","unstructured":"A Acero, L Deng, T Kristjansson, J Zhang, in Proc. ICSLP \u201900. HMM adaptation using vector Taylor series for noisy speech recognition (Beijing, China, 2000), pp. 869\u2013872.","DOI":"10.21437\/ICSLP.2000-672"},{"issue":"3","key":"300_CR40","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1016\/j.csl.2009.02.001","volume":"23","author":"J Li","year":"2009","unstructured":"J Li, L Deng, D Yu, Y Gong, A Acero, A unified framework of HMM adaptation with joint compensation of additive and convolutive distortions. Comput. Speech Lang.23(3), 389\u2013405 (2009).","journal-title":"Comput. Speech Lang."},{"key":"300_CR41","doi-asserted-by":"crossref","unstructured":"Y Li, H Erdogan, Y Gao, E Marcheret, in Proc. ICSLP \u201902. Incremental on-line feature space MLLR adaptation for telephony speech recognition (Denver, USA, 2002), pp. 1417\u20131420.","DOI":"10.21437\/ICSLP.2002-64"},{"issue":"4","key":"300_CR42","doi-asserted-by":"publisher","first-page":"578","DOI":"10.1109\/89.326616","volume":"2","author":"H Hermansky","year":"1994","unstructured":"H Hermansky, N Morgan, RASTA processing of speech. IEEE Trans. Speech Audio Process.2(4), 578\u2013589 (1994).","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"1","key":"300_CR43","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1109\/TASL.2006.876717","volume":"15","author":"C-P Chen","year":"2007","unstructured":"C-P Chen, JA Bilmes, MVA processing of speech features. IEEE Trans. Audio, Speech, Lang. Process.15(1), 257\u2013270 (2007).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"issue":"8","key":"300_CR44","doi-asserted-by":"publisher","first-page":"1662","DOI":"10.1109\/TASL.2008.2002082","volume":"16","author":"X Xiao","year":"2008","unstructured":"X Xiao, ES Chng, H Li, Normalization of the speech modulation spectra for robust speech recognition. IEEE Trans. Audio, Speech, Lang. Process.16(8), 1662\u20131674 (2008).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"key":"300_CR45","volume-title":"Proc. ICASSP \u201913","author":"X Xiao","year":"2013","unstructured":"X Xiao, ES Chng, H Li, in Proc. ICASSP \u201913. Temporal filter design by minimum KL divergence criterion for robust speech recognition (VancouverCanada, 2013)."},{"key":"300_CR46","unstructured":"K Kinoshita, M Delcroix, T Yoshioka, T Nakatani, E Habets, R Haeb-Umbach, V Leutnant, A Sehr, W Kellermann, R Maas, S Gannot, B Raj, in Proceedings of the IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA-13). The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech (New Paltz, NY, 2013)."},{"key":"300_CR47","doi-asserted-by":"crossref","unstructured":"T Robinson, J Fransen, D Pye, J Foote, S Renals, in Proc. ICASSP \u201995. WSJCAM0: a British English speech corpus for large vocabulary continuous speech recognition (Detroit, MI, 1995), pp. 81\u201384.","DOI":"10.1109\/ICASSP.1995.479278"},{"key":"300_CR48","doi-asserted-by":"crossref","unstructured":"DB Paul, JM Baker, in Proceedings of the Workshop on Speech and Natural Language (HLT-91). The design for the wall street journal-based csr corpus (Stroudsburg, PA, 1992), pp. 357\u2013362.","DOI":"10.3115\/1075527.1075614"},{"key":"300_CR49","doi-asserted-by":"crossref","unstructured":"M Lincoln, I McCowan, J Vepa, HK Maganti, in Proc. ASRU \u201905. The multi-channel wall street journal audio visual corpus (MC-WSJ-AV): specification and initial experiments (Cancun, Mexico, 2005), pp. 357\u2013362.","DOI":"10.1109\/ASRU.2005.1566470"},{"issue":"1","key":"300_CR50","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2008","unstructured":"Y Hu, P Loizou, Evaluation of objective quality measures for speech enhancement. IEEE Trans. Audio, Speech, Lang. Process.16(1), 229\u2013238 (2008).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"issue":"7","key":"300_CR51","doi-asserted-by":"publisher","first-page":"1766","DOI":"10.1109\/TASL.2010.2052247","volume":"18","author":"TH Falk","year":"2010","unstructured":"TH Falk, C Zheng, W-Y Chan, A non-intrusive quality and intelligibility measure of reverberant and dereverberated speech. IEEE Trans. Audio, Speech, Lang. Process.18(7), 1766\u20131774 (2010).","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"issue":"10","key":"300_CR52","first-page":"755","volume":"50","author":"A Rix","year":"2002","unstructured":"A Rix, M Hollier, A Hekstra, JG Beerends, Perceptual evaluation of speech quality (PESQ), the new ITU standard for end-to-end speech quality assessment, Part I-time-delay compensation. J. Audio Eng. Soc.50(10), 755\u2013764 (2002).","journal-title":"J. Audio Eng. Soc."},{"key":"300_CR53","unstructured":"D Povey, A Ghoshal, G Boulianne, L Burget, O Glembek, N Goel, M Hannemann, P Motlicek, Y Qian, P Schwarz, J Silovsky, G Stemmer, K Vesely, in Proc. ASRU \u201911. The kaldi speech recognition toolkit (Waikoloa, HI, 2011)."}],"container-title":["EURASIP Journal on Advances in Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13634-015-0300-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1186\/s13634-015-0300-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13634-015-0300-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13634-015-0300-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,1]],"date-time":"2025-06-01T03:45:43Z","timestamp":1748749543000},"score":1,"resource":{"primary":{"URL":"https:\/\/asp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13634-015-0300-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,1,13]]},"references-count":53,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2016,12]]}},"alternative-id":["300"],"URL":"https:\/\/doi.org\/10.1186\/s13634-015-0300-4","relation":{},"ISSN":["1687-6180"],"issn-type":[{"value":"1687-6180","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,1,13]]},"article-number":"4"}}