{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T04:14:43Z","timestamp":1750997683526,"version":"3.41.0"},"publisher-location":"Cham","reference-count":91,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_9","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"219-243","source":"Crossref","is-referenced-by-count":6,"title":["Adaptation of Deep Neural Network Acoustic Models for Robust Automatic Speech Recognition"],"prefix":"10.1007","author":[{"given":"Khe Chai","family":"Sim","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanmin","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gautam","family":"Mantena","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lahiru","family":"Samarakoon","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Souvik","family":"Kundu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tian","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"9_CR1","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid, O., Jiang, H.: Fast speaker adaptation of hybrid NN\/HMM model for speech recognition based on discriminative learning of speaker code. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a07942\u20137946 (2013)","DOI":"10.1109\/ICASSP.2013.6639211"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Abrash, V., Franco, H., Sankar, A., Cohen, M.: Connectionist speaker normalization and adaptation. In: Eurospeech, pp.\u00a02183\u20132186. ISCA (1995)","DOI":"10.21437\/Eurospeech.1995-414"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Chunyang, W., Gales, M.J.: Multi-basis adaptive neural network for rapid adaptation in speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04315\u20134319. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178785"},{"key":"9_CR4","doi-asserted-by":"crossref","unstructured":"Chunyang, W., Karanasou, P., Gales, M.J.: Combining i-vector representation and structured neural networks for rapid adaptation. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a05000\u20135004. IEEE (2016)","DOI":"10.1109\/ICASSP.2016.7472629"},{"issue":"9","key":"9_CR5","doi-asserted-by":"crossref","first-page":"1469","DOI":"10.1109\/TASLP.2015.2438544","volume":"23","author":"X. Cui","year":"2015","unstructured":"Cui, X., Goel, V., Kingsbury, B.: Data augmentation for deep neural network acoustic modeling. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(9), 1469\u20131477 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"9_CR6","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G. Dahl","year":"2012","unstructured":"Dahl, G., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(1), 30\u201342 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"9_CR7","first-page":"1559","volume":"9","author":"N. Dehak","year":"2009","unstructured":"Dehak, N., Dehak, R., Kenny, P., Br\u00fcmmer, N., Ouellet, P., Dumouchel, P.: Support vector machines versus fast scoring in the low-dimensional total variability space for speaker verification. In: Proceedings of Interspeech, vol.\u00a09, pp.\u00a01559\u20131562 (2009)","journal-title":"In: Proceedings of Interspeech"},{"issue":"4","key":"9_CR8","doi-asserted-by":"crossref","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N. Dehak","year":"2011","unstructured":"Dehak, N., Kenny, P., Dehak, R., Dumouchel, P., Ouellet, P.: Front-end factor analysis for speaker verification. IEEE Trans. Audio Speech Lang. Process. 19(4), 788\u2013798 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Delcroix, M., Kinoshita, K., Hori, T., Nakatani, T.: Context adaptive deep neural networks for fast acoustic model adaptation. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04535\u20134539. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178829"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Delcroix, M., Kinoshita, K., Chengzhu, Y., Atsunori, O.: Context adaptive deep neural networks for fast acoustic model adaptation in noise conditions. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a05270\u20135274. IEEE (2016)","DOI":"10.1109\/ICASSP.2016.7472683"},{"issue":"2","key":"9_CR11","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"M.J.F. Gales","year":"1998","unstructured":"Gales, M.J.F.: Maximum likelihood linear transformations for HMM-based speech recognition. Comput. Speech Lang. 12(2), 75\u201398 (1998)","journal-title":"Comput. Speech Lang."},{"issue":"4","key":"9_CR12","doi-asserted-by":"crossref","first-page":"417","DOI":"10.1109\/89.848223","volume":"8","author":"M.J. Gales","year":"2000","unstructured":"Gales, M.J.: Cluster adaptive training of hidden Markov models. IEEE Trans. Speech Audio Process. 8(4), 417\u2013428 (2000)","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"2","key":"9_CR13","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1109\/89.279278","volume":"2","author":"J.L. Gauvain","year":"1994","unstructured":"Gauvain, J.L., Lee, C.H.: Maximum a posteriori estimation for multivariate Gaussian mixture observations of Markov chains. IEEE Trans. Speech Audio Process. 2(2), 291\u2013298 (1994)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Gemello, R., Mana, F., Scanzio, S., Laface, P., Mori, R.D.: Adaptation of hybrid ANN\/HMM models using linear hidden transformations and conservative training. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a01189\u20131192. IEEE (2006)","DOI":"10.1109\/ICASSP.2006.1660239"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Giri, R., Seltzer, M.L., Droppo, J., Yu, D.: Improving speech recognition in reverberation using a room-aware deep neural network and multi-task learning. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a05014\u20135018 (2015)","DOI":"10.1109\/ICASSP.2015.7178925"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Gr\u00e9zl, F., Fousek, P.: Optimizing bottle-neck features for LVCSR. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04729\u20134732 (2008)","DOI":"10.1109\/ICASSP.2008.4518713"},{"key":"9_CR17","first-page":"757","volume":"4","author":"F. Gr\u00e9zl","year":"2007","unstructured":"Gr\u00e9zl, F., Karafiat, M., Kontar, S., Cernocky, J.: Probabilistic and bottle-neck features for LVCSR of meetings. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, vol.\u00a04, pp.\u00a0757\u2013760 (2007)","journal-title":"In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP"},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Gr\u00e9zl, F., Karafi\u00e1t, M., Janda, M.: Study of probabilistic and bottle-neck features in multilingual environment. In: Proceedings of IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a0359\u2013364 (2011)","DOI":"10.1109\/ASRU.2011.6163958"},{"key":"9_CR19","doi-asserted-by":"crossref","unstructured":"Gupta, V., Kenny, P., Ouellet, P., Stafylakis, T.: I-vector-based speaker adaptation of deep neural networks for French broadcast audio transcription. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a06334\u20136338 (2014)","DOI":"10.1109\/ICASSP.2014.6854823"},{"issue":"2","key":"9_CR20","doi-asserted-by":"crossref","first-page":"486","DOI":"10.1109\/TASL.2011.2163395","volume":"20","author":"T. Hain","year":"2012","unstructured":"Hain, T., Burget, L., Dines, J., Garner, P.N., Gr\u00e9zl, F., Hannani, A.E., Huijbregts, M., Karafiat, M., Lincoln, M., Wan, V.: Transcribing meetings with the AMIDA systems. IEEE Trans. Audio Speech Lang. Process. 20(2), 486\u2013498 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"9_CR21","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G. Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G., Mohamed, A., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T.N., Kingsbury, B.: Deep neural networks for acoustic modeling in speech recognition. IEEE Signal Process. Mag. 29, 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"9_CR22","unstructured":"Hinton, G.E., Srivastava, N., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.R.: Improving neural networks by preventing co-adaptation of feature detectors. arXiv:1207.0580 (2012, arXiv preprint)"},{"key":"9_CR23","unstructured":"Hirsch, G.: Experimental framework for the performance evaluation of speech recognition front-ends on a large vocabulary task, version 2.0. ETSI STQ-Aurora DSR Working Group (2002)"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Huang, H., Sim, K.C.: An investigation of augmenting speaker representations to improve speaker normalisation for DNN-based speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04610\u20134613 (2015)","DOI":"10.1109\/ICASSP.2015.7178844"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Ishii, T., Komiyama, H., Shinozaki, T., Horiuchi, Y., Kuroiwa, S.: Reverberant speech recognition based on denoising autoencoder. In: Proceedings of Interspeech, pp.\u00a03512\u20133516 (2013)","DOI":"10.21437\/Interspeech.2013-267"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Karanasou, P., Wang, Y., Gales, M.J.F., Woodland, P.C.: Adaptation of deep neural network acoustic models using factorised i-vectors. In: Proceedings of Interspeech, pp.\u00a02180\u20132184 (2014)","DOI":"10.21437\/Interspeech.2014-488"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Karanasou, P., Gales, M.J.F., Woodland, P.C.: I-vector estimation using informative priors for adaptation of deep neural networks. In: Interspeech, pp.\u00a02872\u20132876 (2015)","DOI":"10.21437\/Interspeech.2015-604"},{"issue":"5","key":"9_CR28","doi-asserted-by":"crossref","first-page":"980","DOI":"10.1109\/TASL.2008.925147","volume":"16","author":"P. Kenny","year":"2008","unstructured":"Kenny, P., Ouellet, P., Dehak, N., Gupta, V., Dumouchel, P.: A study of interspeaker variability in speaker verification. IEEE Trans. Audio Speech Lang. Process. 16(5), 980\u2013988 (2008)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"4","key":"9_CR29","doi-asserted-by":"crossref","first-page":"320","DOI":"10.1109\/TASSP.1976.1162830","volume":"24","author":"C.H. Knapp","year":"1976","unstructured":"Knapp, C.H., Carter, G.C.: The generalized correlation method for estimation of time delay. IEEE Trans. Acoust. Speech Signal Process. 24(4), 320\u2013327 (1976)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"9_CR30","doi-asserted-by":"crossref","unstructured":"Kumar, K., Singh, R., Raj, B., Stern, R.: Gammatone sub-band magnitude-domain dereverberation for ASR. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04604\u20134607. IEEE (2011)","DOI":"10.1109\/ICASSP.2011.5947380"},{"key":"9_CR31","volume-title":"Intermediate-layer DNN adaptation for offline and session-based iterative speaker adaptation","author":"K. Kumar","year":"2015","unstructured":"Kumar, K., Liu, C., Yao, K., Gong, Y.: Intermediate-layer DNN adaptation for offline and session-based iterative speaker adaptation. In: Proceedings of Interspeech. ISCA (2015)"},{"key":"9_CR32","volume-title":"Joint acoustic factor learning for robust deep neural network based automatic speech recognition","author":"S. Kundu","year":"2016","unstructured":"Kundu, S., Mantena, G., Qian, Y., Tan, T., Delcroix, M., Sim, K.C.: Joint acoustic factor learning for robust deep neural network based automatic speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP (2016)"},{"issue":"2","key":"9_CR33","doi-asserted-by":"crossref","first-page":"171","DOI":"10.1006\/csla.1995.0010","volume":"9","author":"C.J. Leggetter","year":"1995","unstructured":"Leggetter, C.J., Woodland, P.C.: Maximum likelihood linear regression for speaker adaptation of continuous density hidden Markov models. Comput. Speech Lang. 9(2), 171\u2013185 (1995)","journal-title":"Comput. Speech Lang."},{"key":"9_CR34","doi-asserted-by":"crossref","unstructured":"Li, B., Sim, K.: Comparison of discriminative input and output transformation for speaker adaptation in the hybrid NN\/HMM systems. In: Proceedings of Interspeech, pp.\u00a0526\u2013529. ISCA (2010)","DOI":"10.21437\/Interspeech.2010-214"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Li, B., Sim, K.C.: Noise adaptive front-end normalization based on vector Taylor series for deep neural networks in robust speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a07408\u20137412. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639102"},{"key":"9_CR36","doi-asserted-by":"crossref","unstructured":"Liao, H.: Speaker adaptation of context dependent deep neural networks. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a07947\u20137951. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639212"},{"key":"9_CR37","doi-asserted-by":"crossref","unstructured":"Lippman, R.P., Martin, E.A., Paul, D.B.: Multi-style training for robust isolated-word speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, vol.\u00a012, pp.\u00a0705\u2013708. IEEE (1987)","DOI":"10.1109\/ICASSP.1987.1169544"},{"issue":"1","key":"9_CR38","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1109\/TASLP.2013.2285487","volume":"22","author":"S. Liu","year":"2014","unstructured":"Liu, S., Sim, K.C.: Temporally varying weight regression: a semi-parametric trajectory model for automatic speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(1), 151\u2013160 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR39","doi-asserted-by":"crossref","unstructured":"Liu, Y., Karanasou, P., Hain, T.: An investigation into speaker informed DNN front-end for LVCSR. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04300\u20134304 (2015)","DOI":"10.1109\/ICASSP.2015.7178782"},{"key":"9_CR40","doi-asserted-by":"crossref","unstructured":"Lu, X., Tsao, Y., Matsuda, S., Hori, C.: Speech enhancement based on deep denoising autoencoder. In: Proceedings of Interspeech, pp.\u00a0436\u2013440 (2013)","DOI":"10.21437\/Interspeech.2013-130"},{"key":"9_CR41","volume-title":"Distance-aware DNNS for robust speech recognition","author":"Y. Miao","year":"2015","unstructured":"Miao, Y., Metze, F.: Distance-aware DNNS for robust speech recognition. In: Proceedings of Interspeech (2015)"},{"key":"9_CR42","doi-asserted-by":"crossref","unstructured":"Miao, Y., Jiang, L., Zhang, H., Metze, F.: Improvements to speaker adaptive training of deep neural networks. In: IEEE Spoken Language Technology Workshop (SLT), 2014, pp.\u00a0165\u2013170. IEEE (2014)","DOI":"10.1109\/SLT.2014.7078568"},{"key":"9_CR43","doi-asserted-by":"crossref","unstructured":"Miao, Y., Zhang, H., Metze, F.: Towards speaker adaptive training of deep neural network acoustic models. In: Proceedings of Interspeech, pp.\u00a02189\u20132193 (2014)","DOI":"10.1109\/SLT.2014.7078568"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Moreno, P.J., Raj, B., Stern, R.M.: A vector Taylor series approach for environment-independent speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, vol.\u00a02, pp.\u00a0733\u2013736. IEEE (1996)","DOI":"10.1109\/ICASSP.1996.543225"},{"key":"9_CR45","volume-title":"Exploring how deep neural networks form phonemic categories","author":"T. Nagamine","year":"2015","unstructured":"Nagamine, T., Seltzer, M.L., Mesgarani, N.: Exploring how deep neural networks form phonemic categories. In: Proceedings of Interspeech (2015)"},{"key":"9_CR46","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-84996-056-4","volume-title":"Speech Dereverberation","author":"P.A. Naylor","year":"2010","unstructured":"Naylor, P.A., Gaubitch, N.D.: Speech Dereverberation. Springer Science & Business Media, London (2010)"},{"key":"9_CR47","volume-title":"Speaker-adaptation for hybrid HMM-ANN continuous speech recognition system","author":"J. Neto","year":"1995","unstructured":"Neto, J., Almeida, L., Hochberg, M., Martins, C., Nunes, L., Renals, S., Robinson, T.: Speaker-adaptation for hybrid HMM-ANN continuous speech recognition system. In: Proceedings of Interspeech. ISCA (1995)"},{"key":"9_CR48","volume-title":"Reverberation robust acoustic modeling using i-vectors with time delay neural networks","author":"V. Peddinti","year":"2015","unstructured":"Peddinti, V., Chen, G., Povey, D., Khudanpur, S.: Reverberation robust acoustic modeling using i-vectors with time delay neural networks. In: Proceedings of Interspeech (2015)"},{"key":"9_CR49","doi-asserted-by":"crossref","unstructured":"Qian, Y., Yin, M., You, Y., Yu, K.: Multi-task joint-learning of deep neural networks for robust speech recognition. In: Proceedings of IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), Scottsdale, AZ, pp.\u00a0310\u2013316 (2015)","DOI":"10.1109\/ASRU.2015.7404810"},{"key":"9_CR50","doi-asserted-by":"crossref","unstructured":"Qian, Y., Tan, T., Yu, D.: An investigation into using parallel data for far-field speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, Shanghai, China, pp.\u00a05725\u20135729 (2016)","DOI":"10.1109\/ICASSP.2016.7472774"},{"key":"9_CR51","doi-asserted-by":"crossref","unstructured":"Qian, Y., Tan, T., Yu, D., Zhang, Y.: Integrated adaptation with multi-factor joint-learning for far-field speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, Shanghai, pp.\u00a05770\u20135774 (2016)","DOI":"10.1109\/ICASSP.2016.7472783"},{"issue":"2","key":"9_CR52","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1109\/5.18626","volume":"77","author":"L.R. Rabiner","year":"1989","unstructured":"Rabiner, L.R.: A tutorial on hidden Markov models and selected applications in speech recognition. Proc. IEEE 77(2), 257\u2013286 (1989)","journal-title":"Proc. IEEE"},{"key":"9_CR53","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Kingsbury, B., Sindhwani, V., Arisoy, E., Ramabhadran, B.: Low-rank matrix factorization for deep neural network training with high-dimensional output targets. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a06655\u20136659. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6638949"},{"issue":"12","key":"9_CR54","doi-asserted-by":"crossref","first-page":"2241","DOI":"10.1109\/TASLP.2016.2601146","volume":"24","author":"L. Samarakoon","year":"2016","unstructured":"Samarakoon, L., Sim, K.C.: Factorized hidden layer adaptation for deep neural network based acoustic modeling. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(12), 2241\u20132250 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR55","volume-title":"On combining i-vectors and discriminative adaptation methods for unsupervised speaker normalisation in DNN acoustic models","author":"L. Samarakoon","year":"2016","unstructured":"Samarakoon, L., Sim, K.C.: On combining i-vectors and discriminative adaptation methods for unsupervised speaker normalisation in DNN acoustic models. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP (2016)"},{"key":"9_CR56","volume-title":"Subspace LHUC for fast adaptation of deep neural network acoustic models","author":"L. Samarakoon","year":"2016","unstructured":"Samarakoon, L., Sim, K.C.: Subspace LHUC for fast adaptation of deep neural network acoustic models. In: Interspeech (2016)"},{"key":"9_CR57","doi-asserted-by":"crossref","unstructured":"Saon, G., Soltau, H., Nahamoo, D., Picheny, M.: Speaker adaptation of neural network acoustic models using i-vectors. In: Proceedings of IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a055\u201359 (2013)","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"9_CR58","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: Proceedings of IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a024\u201329. IEEE (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"9_CR59","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Yu, D.: Conversational speech transcription using context-dependent deep neural networks. In: Proceedings of Interspeech, pp.\u00a0437\u2013440 (2011)","DOI":"10.21437\/Interspeech.2011-169"},{"key":"9_CR60","doi-asserted-by":"crossref","unstructured":"Seltzer, M.L., Yu, D., Wang, Y.: An investigation of deep neural networks for noise robust speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a07398\u20137402 (2013)","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"9_CR61","doi-asserted-by":"crossref","unstructured":"Senior, A., Moreno, I.L.: Improving DNN speaker independence with i-vector inputs. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a0225\u2013229 (2014)","DOI":"10.1109\/ICASSP.2014.6853591"},{"key":"9_CR62","doi-asserted-by":"crossref","unstructured":"Shaofei, X., Abdel-Hamid, O., Hui, J., Lirong, D.: Direct adaptation of hybrid DNN\/HMM model for fast speaker adaptation in LVCSR based on speaker code. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a06339\u20136343. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6854824"},{"issue":"12","key":"9_CR63","doi-asserted-by":"crossref","first-page":"1713","DOI":"10.1109\/TASLP.2014.2346313","volume":"22","author":"X. Shaofei","year":"2014","unstructured":"Shaofei, X., Abdel-Hamid, O., Hui, J., Lirong, D., Qingfeng, L.: Fast adaptation of deep neural network based on discriminant codes for speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(12), 1713\u20131725 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR64","volume-title":"Joint adaptation and adaptive training of TVWR for robust automatic speech recognition","author":"L. Shilin","year":"2014","unstructured":"Shilin, L., Sim, K.C.: Joint adaptation and adaptive training of TVWR for robust automatic speech recognition. In: Proceedings of Interspeech (2014)"},{"key":"9_CR65","doi-asserted-by":"crossref","unstructured":"Sim, K.C.: On constructing and analysing an interpretable brain model for the DNN based on hidden activity patterns. In: Proceedings of Automatic Speech Recognition and Understanding (ASRU), pp.\u00a022\u201329 (2015)","DOI":"10.1109\/ASRU.2015.7404769"},{"key":"9_CR66","doi-asserted-by":"crossref","unstructured":"Stadermann, J., Rigoll, G.: Two-stage speaker adaptation of hybrid tied-posterior acoustic models. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a0977\u2013980 (2005)","DOI":"10.1109\/ICASSP.2005.1415279"},{"key":"9_CR67","doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Renals, S.: Learning hidden unit contributions for unsupervised speaker adaptation of neural network acoustic models. In: Proceedings of IEEE Spoken Language Technology Workshop (SLT), pp.\u00a0171\u2013176. IEEE (2014)","DOI":"10.1109\/SLT.2014.7078569"},{"key":"9_CR68","volume-title":"SAT-LHUC: speaker adaptive training for learning hidden unit contributions","author":"P. Swietojanski","year":"2016","unstructured":"Swietojanski, P., Renals, S.: SAT-LHUC: speaker adaptive training for learning hidden unit contributions. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP. IEEE (2016)"},{"key":"9_CR69","doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Hybrid acoustic models for distant and multichannel large vocabulary speech recognition. In: Proceedings of IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a0285\u2013290 (2013)","DOI":"10.1109\/ASRU.2013.6707744"},{"key":"9_CR70","doi-asserted-by":"crossref","unstructured":"Tan, S., Sim, K.C., Gales, M.: Improving the interpretability of deep neural networks with stimulated learning. In: Proceedings of Automatic Speech Recognition and Understanding (ASRU), pp.\u00a0617\u2013623 (2015)","DOI":"10.1109\/ASRU.2015.7404853"},{"key":"9_CR71","doi-asserted-by":"crossref","unstructured":"Tan, T., Qian, Y., Yin, M., Zhuang, Y., Yu, K.: Cluster adaptive training for deep neural network. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, Brisbane, pp.\u00a04325\u20134329 (2015)","DOI":"10.1109\/ICASSP.2015.7178787"},{"issue":"03","key":"9_CR72","doi-asserted-by":"crossref","first-page":"459","DOI":"10.1109\/TASLP.2015.2511922","volume":"24","author":"T. Tan","year":"2016","unstructured":"Tan, T., Qian, Y., Yu, K.: Cluster adaptive training for deep neural network based acoustic model. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(03), 459\u2013468 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR73","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1007\/978-3-642-15760-8_54","volume-title":"Text, Speech and Dialogue","author":"J. Trmal","year":"2010","unstructured":"Trmal, J., Zelinka, J., M\u00fcller, L.: Adaptation of a feedforward artificial neural network using a linear transform. In: Sojka, P., et al. (eds.) Text, Speech and Dialogue, pp.\u00a0423\u2013430. Springer, Berlin\/Heidelberg (2010)"},{"key":"9_CR74","doi-asserted-by":"crossref","unstructured":"Variani, E., McDermott, E., Heigold, G.: A Gaussian mixture model layer jointly optimized with discriminative features within a deep neural network architecture. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a04270\u20134274. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178776"},{"key":"9_CR75","doi-asserted-by":"crossref","unstructured":"Vesely, K., Karafiat, M., Grezl, F., Janda, M., Egorova, E.: The language-independent bottleneck features. In: Proceedings of IEEE Spoken Language Technology Workshop (SLT), pp.\u00a0336\u2013341 (2012)","DOI":"10.1109\/SLT.2012.6424246"},{"key":"9_CR76","unstructured":"Vu, N.T., Metze, F., Schultz, T.: Multilingual bottle-neck features and its application for under-resourced languages. In: Proceedings of Workshop on Spoken Language Technologies for Under-Resourced Languages (SLTU), pp.\u00a090\u201393 (2012)"},{"key":"9_CR77","doi-asserted-by":"crossref","unstructured":"Wu, C., Karanasou, P., Gales, M.J., Sim, K.C.: Stimulated deep neural network for speech recognition. In: Proceedings of Interspeech, pp.\u00a0400\u2013404. ISCA (2016)","DOI":"10.21437\/Interspeech.2016-580"},{"key":"9_CR78","volume-title":"A initial attempt on task-specific adaptation for deep neural network-based large vocabulary continuous speech recognition","author":"Y. Xiao","year":"2012","unstructured":"Xiao, Y., Zhang, Z., Cai, S., Pan, J., Yan, Y.: A initial attempt on task-specific adaptation for deep neural network-based large vocabulary continuous speech recognition. In: Proceedings of Interspeech. ISCA (2012)"},{"issue":"1","key":"9_CR79","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y. Xu","year":"2014","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: An experimental study on speech enhancement based on deep neural networks. IEEE Signal Process. Lett. 21(1), 65\u201368 (2014)","journal-title":"IEEE Signal Process. Lett."},{"issue":"1","key":"9_CR80","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","volume":"23","author":"Y. Xu","year":"2015","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: A regression approach to speech enhancement based on deep neural networks. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(1), 7\u201319 (2015)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR81","doi-asserted-by":"crossref","unstructured":"Xue, J., Li, J., Gong, Y.: Restructuring of deep neural network acoustic models with singular value decomposition. In: Proceedings of Interspeech, pp.\u00a02365\u20132369. ISCA (2013)","DOI":"10.21437\/Interspeech.2013-552"},{"key":"9_CR82","doi-asserted-by":"crossref","unstructured":"Xue, J., Li, J., Yu, D., Seltzer, M., Gong, Y.: Singular value decomposition based low-footprint speaker adaptation and personalization for deep neural network. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing ICASSP, pp.\u00a06359\u20136363. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6854828"},{"key":"9_CR83","doi-asserted-by":"crossref","unstructured":"Xue, S., Abdel-Hamid, O., Jiang, H., Dai, L.: Direct adaptation of hybrid DNN\/HMM model for fast speaker adaptation in LVCSR based on speaker code. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a06339\u20136343 (2014)","DOI":"10.1109\/ICASSP.2014.6854824"},{"issue":"12","key":"9_CR84","doi-asserted-by":"crossref","first-page":"1713","DOI":"10.1109\/TASLP.2014.2346313","volume":"22","author":"S. Xue","year":"2014","unstructured":"Xue, S., Abdel-Hamid, O., Jiang, H., Dai, L., Liu, Q.: Fast adaptation of deep neural network based on discriminant codes for speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(12), 1713\u20131725 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR85","doi-asserted-by":"crossref","unstructured":"Xue, S., Jiang, H., Dai, L.: Speaker adaptation of hybrid NN\/HMM model for speech recognition based on singular value decomposition. In: ISCSLP, pp.\u00a01\u20135. IEEE (2014)","DOI":"10.1109\/ISCSLP.2014.6936583"},{"issue":"12","key":"9_CR86","doi-asserted-by":"crossref","first-page":"2231","DOI":"10.1109\/TASLP.2016.2598308","volume":"24","author":"T.T. Yanmin Qian","year":"2016","unstructured":"Yanmin\u00a0Qian, T.T., Yu, D.: Neural network based multi-factor aware joint training for robust speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(12), 2231\u20132240 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"9_CR87","doi-asserted-by":"crossref","unstructured":"Yao, K., Yu, D., Seide, F., Su, H., Deng, L., Gong, Y.: Adaptation of context-dependent deep neural networks for automatic speech recognition. In: IEEE Spoken Language Technology Workshop (SLT), 2012, pp.\u00a0366\u2013369. IEEE (2012)","DOI":"10.1109\/SLT.2012.6424251"},{"issue":"6","key":"9_CR88","doi-asserted-by":"crossref","first-page":"114","DOI":"10.1109\/MSP.2012.2205029","volume":"29","author":"T. Yoshioka","year":"2012","unstructured":"Yoshioka, T., Sehr, A., Delcroix, M., Kinoshita, K., Maas, R., Nakatani, T., Kellermann, W.: Making machines understand us in reverberant rooms: robustness against reverberation for automatic speech recognition. IEEE Signal Process. Mag. 29(6), 114\u2013126 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"9_CR89","volume-title":"Automatic Speech Recognition: A Deep Learning Approach","author":"D. Yu","year":"2014","unstructured":"Yu, D., Deng, L.: Automatic Speech Recognition: A Deep Learning Approach. Springer, London (2014)"},{"key":"9_CR90","doi-asserted-by":"crossref","unstructured":"Yu, D., Yao, K., Su, H., Li, G., Seide, F.: KL-divergence regularized deep neural network adaptation for improved large vocabulary speech recognition. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a07893\u20137897. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639201"},{"key":"9_CR91","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Yu, D., Seltzer, M.L., Droppo, J.: Speech recognition with prediction\u2013adaptation\u2013correction recurrent neural networks. In: Proceedings of IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, pp.\u00a05004\u20135008 (2015)","DOI":"10.1109\/ICASSP.2015.7178923"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T20:16:00Z","timestamp":1750968960000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":91,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_9","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}