{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T02:44:12Z","timestamp":1773024252751,"version":"3.50.1"},"publisher-location":"Cham","reference-count":123,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319646794","type":"print"},{"value":"9783319646800","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_8","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"187-217","source":"Crossref","is-referenced-by-count":24,"title":["Robust Features in Deep-Learning-Based Speech Recognition"],"prefix":"10.1007","author":[{"given":"Vikramjit","family":"Mitra","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Horacio","family":"Franco","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Richard M.","family":"Stern","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Julien","family":"van Hout","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luciana","family":"Ferrer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Martin","family":"Graciarena","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dimitra","family":"Vergyri","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abeer","family":"Alwan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John H. L.","family":"Hansen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"8_CR1","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid, O., Mohamed, A.R., Jiang, H., Penn, G.: Applying convolutional neural networks concepts to hybrid NN-HMM model for speech recognition. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4277\u20134280. IEEE (2012)","DOI":"10.1109\/ICASSP.2012.6288864"},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid, O., Deng, L., Yu, D.: Exploring convolutional neural network structures and optimization techniques for speech recognition. In: Interspeech, pp. 3366\u20133370 (2013)","DOI":"10.21437\/Interspeech.2013-744"},{"issue":"2B","key":"8_CR3","doi-asserted-by":"crossref","first-page":"637","DOI":"10.1121\/1.1912679","volume":"50","author":"B.S. Atal","year":"1971","unstructured":"Atal, B.S., Hanauer, S.L.: Speech analysis and synthesis by linear prediction of the speech wave. J. Acoust. Soc. Am. 50(2B), 637\u2013655 (1971)","journal-title":"J. Acoust. Soc. Am."},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Athineos, M., Ellis, D.P.: Frequency-domain linear prediction for temporal features. In: 2003 IEEE Workshop on Automatic Speech Recognition and Understanding, ASRU\u201930, pp. 261\u2013266. IEEE (2003)","DOI":"10.1109\/ASRU.2003.1318451"},{"key":"8_CR5","volume-title":"LP-TRAP: linear predictive temporal patterns","author":"M. Athineos","year":"2004","unstructured":"Athineos, M., Hermansky, H., Ellis, D.P.: LP-TRAP: linear predictive temporal patterns. Technical Report, IDIAP (2004)"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Barker, J., Marxer, R., Vincent, E., Watanabe, S.: The third \u201cchiME\u201d speech separation and recognition challenge: dataset, task and baselines. In: 2015 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU 2015) (2015)","DOI":"10.1109\/ASRU.2015.7404837"},{"key":"8_CR7","volume-title":"Toward human-assisted lexical unit discovery without text resources","author":"C. Bartels","year":"2016","unstructured":"Bartels, C., Wang, W., Mitra, V., Richey, C., Kathol, A., Vergyri, D., Bratt, H., Hung, C.: Toward human-assisted lexical unit discovery without text resources. In: SLT (2016)"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Beh, J., Ko, H.: A novel spectral subtraction scheme for robust speech recognition: spectral subtraction using spectral harmonics of speech. In: 2003 IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP\u201903, vol. 1, pp. I\u2013648. IEEE (2003)","DOI":"10.1007\/3-540-44864-0_115"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Bell, P., Gales, M., Hain, T., Kilgour, J., Lanchantin, P., Liu, X., McParland, A., Renals, S., Saz, O., Wester, M., et al.: The MGB challenge: evaluating multi-genre broadcast media recognition. In: 2015 Automatic Speech Recognition and Understanding Workshop (ASRU 2013) (2015)","DOI":"10.1109\/ASRU.2015.7404863"},{"key":"8_CR10","volume-title":"Speech Enhancement","author":"J. Benesty","year":"2005","unstructured":"Benesty, J., Makino, S.: Speech Enhancement. Springer Science & Business Media, New York (2005)"},{"key":"8_CR11","first-page":"19","volume":"7","author":"Y. Bengio","year":"2012","unstructured":"Bengio, Y.: Deep learning of representations for unsupervised and transfer learning. In: Unsupervised and Transfer Learning Challenges in Machine Learning, vol. 7, p. 19 (2012)","journal-title":"In: Unsupervised and Transfer Learning Challenges in Machine Learning"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Bhargava, M., Rose, R.: Architectures for deep neural network based acoustic models defined over windowed speech waveforms. In: 16th Annual Conference of the International Speech Communication Association (2015)","DOI":"10.21437\/Interspeech.2015-2"},{"issue":"2","key":"8_CR13","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1109\/TASSP.1979.1163209","volume":"27","author":"S.F. Boll","year":"1979","unstructured":"Boll, S.F.: Suppression of acoustic noise in speech using spectral subtraction. IEEE Trans. Acoust. Speech Signal Process. 27(2), 113\u2013120 (1979)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"8_CR14","volume-title":"Auditory Scene Analysis: The Perceptual Organization of Sound","author":"A.S. Bregman","year":"1994","unstructured":"Bregman, A.S.: Auditory Scene Analysis: The Perceptual Organization of Sound. MIT Press, Cambridge, MA (1994)"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Chang, S.Y., Morgan, N.: Robust CNN-based speech recognition with Gabor filter kernels. In: Interspeech, pp. 905\u2013909 (2014)","DOI":"10.21437\/Interspeech.2014-226"},{"key":"8_CR16","first-page":"69","volume":"4","author":"C. Cieri","year":"2004","unstructured":"Cieri, C., Miller, D., Walker, K.: The fisher corpus: a resource for the next generations of speech-to-text. In: LREC, vol. 4, pp. 69\u201371 (2004)","journal-title":"In: LREC"},{"issue":"6","key":"8_CR17","doi-asserted-by":"crossref","first-page":"2623","DOI":"10.1121\/1.397756","volume":"85","author":"J.R. Cohen","year":"1989","unstructured":"Cohen, J.R.: Application of an auditory model to speech recognition. J. Acoust. Soc. Am. 85(6), 2623\u20132629 (1989)","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"8_CR18","doi-asserted-by":"crossref","first-page":"267","DOI":"10.1016\/S0167-6393(00)00034-0","volume":"34","author":"M. Cooke","year":"2001","unstructured":"Cooke, M., Green, P., Josifovski, L., Vizinho, A.: Robust automatic speech recognition with missing and unreliable acoustic data. Speech Commun. 34(3), 267\u2013285 (2001)","journal-title":"Speech Commun."},{"issue":"4","key":"8_CR19","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S.B. Davis","year":"1980","unstructured":"Davis, S.B., Mermelstein, P.: Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Trans. Acoust. Speech Signal Process. 28(4), 357\u2013366 (1980)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"6","key":"8_CR20","doi-asserted-by":"crossref","first-page":"637","DOI":"10.1121\/1.1906946","volume":"24","author":"K. Davis","year":"1952","unstructured":"Davis, K., Biddulph, R., Balashek, S.: Automatic recognition of spoken digits. J. Acoust. Soc. Am. 24(6), 637\u2013642 (1952)","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"8_CR21","doi-asserted-by":"crossref","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N. Dehak","year":"2011","unstructured":"Dehak, N., Kenny, P., Dehak, R., Dumouchel, P., Ouellet, P.: Front-end factor analysis for speaker verification. IEEE Trans. Audio Speech Lang. Process. 19(4), 788\u2013798 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"8_CR22","volume-title":"Linear prediction-based dereverberation with advanced speech enhancement and recognition technologies for the REVERB challenge","author":"M. Delcroix","year":"2014","unstructured":"Delcroix, M., Yoshioka, T., Ogawa, A., Kubo, Y., Fujimoto, M., Ito, N., Kinoshita, K., Espi, M., Hori, T., Nakatani, T., et al.: Linear prediction-based dereverberation with advanced speech enhancement and recognition technologies for the REVERB challenge. In: Proceedings of the REVERB Workshop (2014)"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Deng, L., Hinton, G., Kingsbury, B.: New types of deep neural network learning for speech recognition and related applications: an overview. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8599\u20138603. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639344"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Dennis, J., Dat, T.H.: Single and multi-channel approaches for distant speech recognition under noisy reverberant conditions: I2R\u2019s system description for the ASpIRE challenge. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 518\u2013524. IEEE (2015)","DOI":"10.1109\/ASRU.2015.7404839"},{"issue":"5","key":"8_CR25","doi-asserted-by":"crossref","first-page":"2670","DOI":"10.1121\/1.409836","volume":"95","author":"R. Drullman","year":"1994","unstructured":"Drullman, R., Festen, J.M., Plomp, R.: Effect of reducing slow temporal modulations on speech reception. J. Acoust. Soc. Am. 95(5), 2670\u20132680 (1994)","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"8_CR26","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pcbi.1000302","volume":"5","author":"T.M. Elliott","year":"2009","unstructured":"Elliott, T.M., Theunissen, F.E.: The modulation transfer function for speech intelligibility. PLoS Comput. Biol. 5(3), e1000302 (2009)","journal-title":"PLoS Comput. Biol."},{"key":"8_CR27","unstructured":"ETSI: Speech processing, transmission and quality aspects (STQ); distributed speech recognition; front-end feature extraction algorithm; compression algorithms. ETSI ES 21 108, ver. 1.1.3 (2003)"},{"key":"8_CR28","unstructured":"ETSI: Speech processing, transmission and quality aspects (STQ); distributed speech recognition; advanced front-end feature extraction algorithm; compression algorithms. ETSI ES 202, 050, ver. 1.1.5 (2007)"},{"key":"8_CR29","doi-asserted-by":"crossref","unstructured":"Fine, S., Saon, G., Gopinath, R.A.: Digit recognition in noisy environments via a sequential GMM\/SVM system. In: 2002 IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), vol. 1, pp. I\u201349. IEEE (2002)","DOI":"10.1109\/ICASSP.2002.1005672"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Fiscus, J.G.: A post-processing system to yield reduced word error rates: recognizer output voting error reduction (ROVER). In: 1997 IEEE Workshop on Automatic Speech Recognition and Understanding, pp. 347\u2013354. IEEE (1997)","DOI":"10.1109\/ASRU.1997.659110"},{"issue":"10","key":"8_CR31","doi-asserted-by":"crossref","first-page":"797","DOI":"10.1016\/j.specom.2008.05.004","volume":"50","author":"R. Flynn","year":"2008","unstructured":"Flynn, R., Jones, E.: Combined speech enhancement and auditory modelling for robust distributed speech recognition. Speech Commun. 50(10), 797\u2013809 (2008)","journal-title":"Speech Commun."},{"issue":"4","key":"8_CR32","doi-asserted-by":"crossref","first-page":"249","DOI":"10.1006\/csla.1996.0013","volume":"10","author":"M.J. Gales","year":"1996","unstructured":"Gales, M.J., Woodland, P.C.: Mean and variance adaptation within the MLLR framework. Comput. Speech Lang. 10(4), 249\u2013264 (1996)","journal-title":"Comput. Speech Lang."},{"issue":"6","key":"8_CR33","doi-asserted-by":"crossref","first-page":"3769","DOI":"10.1121\/1.3504658","volume":"128","author":"S. Ganapathy","year":"2010","unstructured":"Ganapathy, S., Thomas, S., Hermansky, H.: Temporal envelope compensation for robust phoneme recognition using modulation spectrum. J. Acoust. Soc. Am. 128(6), 3769\u20133780 (2010)","journal-title":"J. Acoust. Soc. Am."},{"key":"8_CR34","volume-title":"Robust i-vector based adaptation of DNN acoustic model for speech recognition","author":"S. Garimella","year":"2015","unstructured":"Garimella, S., Mandal, A., Strom, N., Hoffmeister, B., Matsoukas, S., Parthasarathi, S.H.K.: Robust i-vector based adaptation of DNN acoustic model for speech recognition. In: Interspeech (2015)"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Gelly, G., Gauvain, J-L.: Minimum word error training of RNN-based voice activity detection. In: Interspeech, pp. 2650\u20132654 (2015)","DOI":"10.21437\/Interspeech.2015-565"},{"key":"8_CR36","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., Virtanen, T.: Noise robust exemplar-based connected digit recognition. In: 2010 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP), pp. 4546\u20134549. IEEE (2010)","DOI":"10.1109\/ICASSP.2010.5495580"},{"issue":"2","key":"8_CR37","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1016\/S0885-2308(86)80018-3","volume":"1","author":"O. Ghitza","year":"1986","unstructured":"Ghitza, O.: Auditory nerve representation as a front-end for speech recognition in a noisy environment. Comput. Speech Lang. 1(2), 109\u2013130 (1986)","journal-title":"Comput. Speech Lang."},{"issue":"3","key":"8_CR38","doi-asserted-by":"crossref","first-page":"1628","DOI":"10.1121\/1.1396325","volume":"110","author":"O. Ghitza","year":"2001","unstructured":"Ghitza, O.: On the upper cutoff frequency of the auditory critical-band envelope detectors in the context of speech perception. J. Acoust. Soc. Am. 110(3), 1628\u20131640 (2001)","journal-title":"J. Acoust. Soc. Am."},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Gibson, J., Van Segbroeck, M., Narayanan, S.S.: Comparing time\u2013frequency representations for directional derivative features. In: Interspeech, pp. 612\u2013615 (2014)","DOI":"10.21437\/Interspeech.2014-147"},{"key":"8_CR40","doi-asserted-by":"crossref","DOI":"10.1017\/CBO9781139166126","volume-title":"English Phonology: An Introduction","author":"H.J. Giegerich","year":"1992","unstructured":"Giegerich, H.J.: English Phonology: An Introduction. Cambridge University Press, Cambridge (1992)"},{"key":"8_CR41","doi-asserted-by":"crossref","unstructured":"Graciarena, M., Alwan, A., Ellis, D., Franco, H., Ferrer, L., Hansen, J.H., Janin, A., Lee, B.S., Lei, Y., Mitra, V., et al.: All for one: feature combination for highly channel-degraded speech activity detection. In: Interspeech, pp. 709\u2013713 (2013)","DOI":"10.21437\/Interspeech.2013-199"},{"key":"8_CR42","doi-asserted-by":"crossref","unstructured":"Graciarena, M., Ferrer, L., Mitra, V.: The SRI system for the NIST open sad 2015 speech activity detection evaluation. In: Interspeech, pp. 3673\u20133677 (2016)","DOI":"10.21437\/Interspeech.2016-550"},{"key":"8_CR43","doi-asserted-by":"crossref","unstructured":"Grezl, F., Egorova, E., Karafi\u00e1t, M.: Further investigation into multilingual training and adaptation of stacked bottle-neck neural network structure. In: 2014 IEEE Spoken Language Technology Workshop (SLT), pp. 48\u201353. IEEE (2014)","DOI":"10.1109\/SLT.2014.7078548"},{"issue":"8","key":"8_CR44","doi-asserted-by":"crossref","first-page":"799","DOI":"10.1109\/89.966083","volume":"9","author":"H. Gustafsson","year":"2001","unstructured":"Gustafsson, H., Nordholm, S.E., Claesson, I.: Spectral subtraction using reduced delay convolution and adaptive averaging. IEEE Trans. Speech Audio Process. 9(8), 799\u2013807 (2001)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"8_CR45","volume-title":"The automatic speech recognition in reverberant environments (ASpIRE) challenge","author":"M. Harper","year":"2015","unstructured":"Harper, M.: The automatic speech recognition in reverberant environments (ASpIRE) challenge. In: ASRU (2015)"},{"key":"8_CR46","doi-asserted-by":"crossref","unstructured":"Harvilla, M.J., Stern, R.M.: Histogram-based subband powerwarping and spectral averaging for robust speech recognition under matched and multistyle training. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4697\u20134700. IEEE (2012)","DOI":"10.1109\/ICASSP.2012.6288967"},{"issue":"4","key":"8_CR47","doi-asserted-by":"crossref","first-page":"1738","DOI":"10.1121\/1.399423","volume":"87","author":"H. Hermansky","year":"1990","unstructured":"Hermansky, H.: Perceptual linear predictive (PLP) analysis of speech. J. Acoust. Soc. Am. 87(4), 1738\u20131752 (1990)","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"8_CR48","doi-asserted-by":"crossref","first-page":"578","DOI":"10.1109\/89.326616","volume":"2","author":"H. Hermansky","year":"1994","unstructured":"Hermansky, H., Morgan, N.: RASTA processing of speech. IEEE Trans. Speech Audio Process. 2(4), 578\u2013589 (1994)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"8_CR49","doi-asserted-by":"crossref","unstructured":"Hermansky, H., Sharma, S.: Temporal patterns (TRAPS) in ASR of noisy speech. In: 1999 IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 1, pp. 289\u2013292. IEEE (1999)","DOI":"10.1109\/ICASSP.1999.758119"},{"issue":"6","key":"8_CR50","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G. Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G.E., Mohamed, A.R., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T.N., et al.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"8_CR51","unstructured":"Hirsch, G.: Experimental framework for the performance evaluation of speech recognition front-ends on a large vocabulary task. ETSI STQ Aurora DSR Working Group (2002)"},{"key":"8_CR52","volume-title":"The MERL\/SRI system for the 3rd CHiME challenge using beamforming, robust feature extraction, and advanced speech recognition","author":"T. Hori","year":"2015","unstructured":"Hori, T., Chen, Z., Erdogan, H., Hershey, J.R., Roux, J., Mitra, V., Watanabe, S.: The MERL\/SRI system for the 3rd CHiME challenge using beamforming, robust feature extraction, and advanced speech recognition. In: Proceedings of the IEEE ASRU (2015)"},{"key":"8_CR53","doi-asserted-by":"crossref","unstructured":"Hoshen, Y., Weiss, R.J., Wilson, K.W.: Speech acoustic modeling from raw multichannel waveforms. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4624\u20134628. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178847"},{"key":"8_CR54","volume-title":"Robust speech recognition in unknown reverberant and noisy conditions","author":"R. Hsiao","year":"2015","unstructured":"Hsiao, R., Ma, J., Hartmann, W., Karafiat, M., Gr\u00e9zl, F., Burget, L., Szoke, I., Cernocky, J., Watanabe, S., Chen, Z., et al.: Robust speech recognition in unknown reverberant and noisy conditions. In: Proceedings of the IEEE Automatic Speech Recognition and Understanding Workshop (2015)"},{"issue":"1","key":"8_CR55","doi-asserted-by":"crossref","first-page":"67","DOI":"10.1109\/TASSP.1975.1162641","volume":"23","author":"F. Itakura","year":"1975","unstructured":"Itakura, F.: Minimum prediction residual principle applied to speech recognition. IEEE Trans. Acoust. Speech Signal Process. 23(1), 67\u201372 (1975)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"issue":"1","key":"8_CR56","first-page":"36","volume":"53","author":"F. Itakura","year":"1970","unstructured":"Itakura, F., Saito, S.: Statistical method for estimation of speech spectral density and formant frequencies. Electron. Commun. Jpn. 53(1), 36 (1970)","journal-title":"Electron. Commun. Jpn."},{"issue":"10","key":"8_CR57","doi-asserted-by":"crossref","first-page":"259","DOI":"10.1109\/97.789604","volume":"6","author":"F. Jabloun","year":"1999","unstructured":"Jabloun, F., Cetin, A.E., Erzin, E.: Teager energy based feature parameters for speech recognition in car noise. IEEE Signal Process. Lett. 6(10), 259\u2013261 (1999)","journal-title":"IEEE Signal Process. Lett."},{"issue":"2","key":"8_CR58","doi-asserted-by":"crossref","first-page":"541","DOI":"10.1152\/physrev.00029.2003","volume":"84","author":"P. Joris","year":"2004","unstructured":"Joris, P., Schreiner, C., Rees, A.: Neural processing of amplitude-modulated sounds. Physiol. Rev. 84(2), 541\u2013577 (2004)","journal-title":"Physiol. Rev."},{"key":"8_CR59","volume-title":"Automatic speech recognition \u2013 a brief history of the technology development","author":"B.H. Juang","year":"2005","unstructured":"Juang, B.H., Rabiner, L.R.: Automatic speech recognition \u2013 a brief history of the technology development. In: Encyclopedia of Language and Linguistics. Elsevier, Amsterdam (2005)"},{"key":"8_CR60","doi-asserted-by":"crossref","unstructured":"Kanedera, N., Arai, T., Hermansky, H., Pavel, M.: On the importance of various modulation frequencies for speech recognition. In: 5th European Conference on Speech Communication and Technology (1997)","DOI":"10.21437\/Eurospeech.1997-104"},{"key":"8_CR61","doi-asserted-by":"crossref","unstructured":"Karafi\u00e1t, M., Gr\u00e9zl, F., Burget, L., Sz\u00f6ke, I., \u010cernock\u1ef3, J.: Three ways to adapt a CTS recognizer to unseen reverberated speech in BUT system for the ASpIRE challenge. In: 16th Annual Conference of the International Speech Communication Association (2015)","DOI":"10.21437\/Interspeech.2015-530"},{"key":"8_CR62","doi-asserted-by":"crossref","unstructured":"Kim, C., Stern, R.M.: Feature extraction for robust speech recognition based on maximizing the sharpness of the power distribution and on power flooring. In: ICASSP, pp. 4574\u20134577 (2010)","DOI":"10.1109\/ICASSP.2010.5495570"},{"key":"8_CR63","unstructured":"Kim, C., Stern, R.M.: Power-Cepstral Coefficients (PNCC) for Robust Speech Recognition. IEEE\/ACM Trans. Audio, Speech, and Language Process. 24(7), 1315\u20131329 (2016)"},{"issue":"1","key":"8_CR64","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/S0167-6393(98)00032-6","volume":"25","author":"B.E. Kingsbury","year":"1998","unstructured":"Kingsbury, B.E., Morgan, N., Greenberg, S.: Robust speech recognition using the modulation spectrogram. Speech Commun. 25(1), 117\u2013132 (1998)","journal-title":"Speech Commun."},{"key":"8_CR65","doi-asserted-by":"crossref","unstructured":"Kingsbury, B., Saon, G., Mangu, L., Padmanabhan, M., Sarikaya, R.: Robust speech recognition in noisy environments: the 2001 IBM spine evaluation system. In: 2002 IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), vol. 1, pp. I\u201353. IEEE (2002)","DOI":"10.1109\/ICASSP.2002.1005673"},{"key":"8_CR66","doi-asserted-by":"crossref","unstructured":"Kingsbury, B., Sainath, T.N., Soltau, H.: Scalable minimum Bayes risk training of deep neural network acoustic models using distributed Hessian-free optimization. In: 13th Annual Conference of the International Speech Communication Association (2012)","DOI":"10.21437\/Interspeech.2012-3"},{"key":"8_CR67","doi-asserted-by":"crossref","unstructured":"Kinoshita, K., Delcroix, M., Yoshioka, T., Nakatani, T., Sehr, A., Kellermann, W., Maas, R.: The REVERB challenge: a common evaluation framework for dereverberation and recognition of reverberant speech. In: 2013 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), pp. 1\u20134. IEEE (2013)","DOI":"10.1109\/WASPAA.2013.6701894"},{"key":"8_CR68","doi-asserted-by":"crossref","unstructured":"Li, X., Bilmes, J.: Regularized adaptation of discriminative classifiers. In: 2006 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2006, vol. 1, pp. I-237\u2013I-240. IEEE (2006)","DOI":"10.1109\/ICASSP.2006.1660001"},{"key":"8_CR69","doi-asserted-by":"crossref","unstructured":"Lyon, R.F.: A computational model of filtering, detection, and compression in the cochlea. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP\u201982, vol. 7, pp. 1282\u20131285. IEEE (1982)","DOI":"10.1109\/ICASSP.1982.1171644"},{"key":"8_CR70","doi-asserted-by":"crossref","unstructured":"Makhoul, J., Cosell, L.: LPCW: an LPC vocoder with linear predictive spectral warping. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP\u201976, vol. 1, pp. 466\u2013469. IEEE (1976)","DOI":"10.1109\/ICASSP.1976.1170013"},{"issue":"10","key":"8_CR71","doi-asserted-by":"crossref","first-page":"3024","DOI":"10.1109\/78.277799","volume":"41","author":"P. Maragos","year":"1993","unstructured":"Maragos, P., Kaiser, J.F., Quatieri, T.F.: Energy separation in signal modulations with application to speech analysis. IEEE Trans. Signal Process. 41(10), 3024\u20133051 (1993)","journal-title":"IEEE Trans. Signal Process."},{"key":"8_CR72","doi-asserted-by":"crossref","unstructured":"Mesgarani, N., David, S., Shamma, S.: Representation of phonemes in primary auditory cortex: how the brain analyzes speech. In: 2007, IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2007, vol. 4, pp. IV\u2013765. IEEE (2007)","DOI":"10.1109\/ICASSP.2007.367025"},{"key":"8_CR73","doi-asserted-by":"crossref","unstructured":"Meyer, B.T., Ravuri, S.V., Sch\u00e4dler, M.R., Morgan, N.: Comparing different flavors of spectro-temporal features for ASR. In: Interspeech, pp. 1269\u20131272 (2011)","DOI":"10.21437\/Interspeech.2011-103"},{"key":"8_CR74","doi-asserted-by":"crossref","unstructured":"Mitra, V., Franco, H.: Time\u2013frequency convolutional networks for robust speech recognition. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 317\u2013323. IEEE (2015)","DOI":"10.1109\/ASRU.2015.7404811"},{"key":"8_CR75","doi-asserted-by":"crossref","unstructured":"Mitra, V., Franco, H.: Coping with unseen data conditions: investigating neural net architectures, robust features, and information fusion for robust speech recognition. In: Interspeech, pp. 3783\u20133787 (2016)","DOI":"10.21437\/Interspeech.2016-966"},{"issue":"6","key":"8_CR76","doi-asserted-by":"crossref","first-page":"1027","DOI":"10.1109\/JSTSP.2010.2076013","volume":"4","author":"V. Mitra","year":"2010","unstructured":"Mitra, V., Nam, H., Espy-Wilson, C.Y., Saltzman, E., Goldstein, L.: Retrieving tract variables from acoustics: a comparison of different machine learning strategies. IEEE J. Sel. Top. Signal. Process. 4(6), 1027\u20131045 (2010)","journal-title":"IEEE J. Sel. Top. Signal. Process."},{"key":"8_CR77","doi-asserted-by":"crossref","unstructured":"Mitra, V., Franco, H., Graciarena, M., Mandal, A.: Normalized amplitude modulation features for large vocabulary noise-robust speech recognition. In: 2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4117\u20134120. IEEE (2012)","DOI":"10.1109\/ICASSP.2012.6288824"},{"key":"8_CR78","doi-asserted-by":"crossref","unstructured":"Mitra, V., Franco, H., Graciarena, M.: Damped oscillator cepstral coefficients for robust speech recognition. In: Interspeech, pp. 886\u2013890 (2013)","DOI":"10.1109\/ICASSP.2014.6853898"},{"key":"8_CR79","doi-asserted-by":"crossref","unstructured":"Mitra, V., Franco, H., Graciarena, M., Vergyri, D.: Medium-duration modulation cepstral feature for robust speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1749\u20131753. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6853898"},{"key":"8_CR80","doi-asserted-by":"crossref","unstructured":"Mitra, V., Wang, W., Franco, H.: Deep convolutional nets and robust features for reverberation-robust speech recognition. In: Spoken Language Technology Workshop (SLT), pp. 548\u2013553. IEEE (2014)","DOI":"10.1109\/SLT.2014.7078633"},{"key":"8_CR81","doi-asserted-by":"crossref","unstructured":"Mitra, V., Wang, W., Franco, H., Lei, Y., Bartels, C., Graciarena, M.: Evaluating robust features on deep neural networks for speech recognition in noisy and channel mismatched conditions. In: Interspeech, pp. 895\u2013899 (2014)","DOI":"10.21437\/Interspeech.2014-224"},{"key":"8_CR82","doi-asserted-by":"crossref","unstructured":"Mitra, V., Hout, J.V., McLaren, M., Wang, W., Graciarena, M., Vergyri, D., Franco, H.: Combating reverberation in large vocabulary continuous speech recognition. In: 16th Annual Conference of the International Speech Communication Association (2015)","DOI":"10.21437\/Interspeech.2015-529"},{"key":"8_CR83","doi-asserted-by":"crossref","unstructured":"Mitra, V., Van Hout, J., Wang, W., Graciarena, M., McLaren, M., Franco, H., Vergyri, D.: Improving robustness against reverberation for automatic speech recognition. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 525\u2013532. IEEE (2015)","DOI":"10.1109\/ASRU.2015.7404840"},{"key":"8_CR84","volume-title":"Fusion strategies for robust speech recognition and keyword spotting for channel- and noise-degraded speech","author":"V. Mitra","year":"2016","unstructured":"Mitra, V., van Hout, J., Wang, W., Bartels, C., Franco, H., Vergyri, D., et al.: Fusion strategies for robust speech recognition and keyword spotting for channel- and noise-degraded speech. In: Interspeech, 2016 (2016)"},{"issue":"1","key":"8_CR85","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A.R. Mohamed","year":"2012","unstructured":"Mohamed, A.R., Dahl, G.E., Hinton, G.: Acoustic modeling using deep belief networks. IEEE Trans. Audio Speech Lang. Process. 20(1), 14\u201322 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"8_CR86","volume-title":"An Introduction to the Psychology of Hearing","author":"B. Moore","year":"1989","unstructured":"Moore, B.: An Introduction to the Psychology of Hearing. Emerald Group Publishing Ltd., Bingley (1989)"},{"issue":"11","key":"8_CR87","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1186\/2190-8567-1-11","volume":"1","author":"A.B. Neiman","year":"2011","unstructured":"Neiman, A.B., Dierkes, K., Lindner, B., Han, L., Shilnikov, A.L., et al.: Spontaneous voltage oscillations and response dynamics of a Hodgkin\u2013Huxley type model of sensory hair cells. J. Math. Neurosci. 1(11), 11 (2011)","journal-title":"J. Math. Neurosci."},{"key":"8_CR88","doi-asserted-by":"crossref","unstructured":"Parthasarathi, S.H.K., Hoffmeister, B., Matsoukas, S., Mandal, A., Strom, N., Garimella, S.: fMLLR based feature-space speaker adaptation of DNN acoustic models. In: 16th Annual Conference of the International Speech Communication Association (2015)","DOI":"10.21437\/Interspeech.2015-720"},{"key":"8_CR89","doi-asserted-by":"crossref","first-page":"429","DOI":"10.1016\/B978-0-08-041847-6.50054-X","volume":"83","author":"R.D. Patterson","year":"1992","unstructured":"Patterson, R.D., Robinson, K., Holdsworth, J., McKeown, D., Zhang, C., Allerhand, M.: Complex sounds and auditory images. Audit. Physiol. Percep. 83, 429\u2013446 (1992)","journal-title":"Audit. Physiol. Percep."},{"key":"8_CR90","volume-title":"JHU ASPIRE system: robust LVCSR with TDNNS, i-vector adaptation, and RNN-LMS","author":"V. Peddinti","year":"2015","unstructured":"Peddinti, V., Chen, G., Manohar, V., Ko, T., Povey, D., Khudanpur, S.: JHU ASPIRE system: robust LVCSR with TDNNS, i-vector adaptation, and RNN-LMS. In: Proceedings of the IEEE Automatic Speech Recognition and Understanding Workshop (2015)"},{"key":"8_CR91","volume-title":"A time delay neural network architecture for efficient modeling of long temporal contexts","author":"V. Peddinti","year":"2015","unstructured":"Peddinti, V., Povey, D., Khudanpur, S.: A time delay neural network architecture for efficient modeling of long temporal contexts. In: Interspeech (2015)"},{"issue":"3","key":"8_CR92","doi-asserted-by":"crossref","first-page":"196","DOI":"10.1109\/89.905994","volume":"9","author":"A. Potamianos","year":"2001","unstructured":"Potamianos, A., Maragos, P.: Time\u2013frequency distributions for automatic speech recognition. IEEE Trans. Speech Audio Process. 9(3), 196\u2013200 (2001)","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"4","key":"8_CR93","doi-asserted-by":"crossref","first-page":"336","DOI":"10.1109\/TASSP.1979.1163259","volume":"27","author":"L.R. Rabiner","year":"1979","unstructured":"Rabiner, L.R., Levinson, S.E., Rosenberg, A.E., Wilpon, J.G.: Speaker-independent recognition of isolated words using clustering techniques. IEEE Trans. Acoust. Speech Signal Process. 27(4), 336\u2013349 (1979)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"8_CR94","doi-asserted-by":"crossref","unstructured":"Rath, S.P., Povey, D., Vesel\u1ef3, K., Cernock\u1ef3, J.: Improved feature processing for deep neural networks. In: Interspeech, pp. 109\u2013113 (2013)","DOI":"10.21437\/Interspeech.2013-48"},{"key":"8_CR95","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Kingsbury, B., Mohamed, A.R., Ramabhadran, B.: Learning filter banks within a deep neural network framework. In: 2013 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 297\u2013302. IEEE (2013)","DOI":"10.1109\/ASRU.2013.6707746"},{"key":"8_CR96","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Mohamed, A.R., Kingsbury, B., Ramabhadran, B.: Deep convolutional neural networks for LVCSR. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8614\u20138618. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639347"},{"key":"8_CR97","volume-title":"Learning the speech front-end with raw waveform CLDNNS","author":"T.N. Sainath","year":"2015","unstructured":"Sainath, T.N., Weiss, R.J., Senior, A., Wilson, K.W., Vinyals, O.: Learning the speech front-end with raw waveform CLDNNS. In: Proceedings of the Interspeech (2015)"},{"key":"8_CR98","doi-asserted-by":"crossref","unstructured":"Saon, G., Soltau, H., Nahamoo, D., Picheny, M.: Speaker adaptation of neural network acoustic models using i-vectors. In: ASRU, pp. 55\u201359 (2013)","DOI":"10.1109\/ASRU.2013.6707705"},{"issue":"324","key":"8_CR99","first-page":"130","volume":"5","author":"M.R. Schroeder","year":"1977","unstructured":"Schroeder, M.R.: Recognition of complex acoustic signals. Life Sci. Res. Rep. 5(324), 130 (1977)","journal-title":"Life Sci. Res. Rep."},{"key":"8_CR100","unstructured":"Schwarz, P.: Phoneme recognition based on long temporal context. Ph.D. thesis, Burno University of Technology (2009)"},{"key":"8_CR101","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: 2011 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 24\u201329. IEEE (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"8_CR102","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Yu, D.: Conversational speech transcription using context-dependent deep neural networks. In: Interspeech, pp. 437\u2013440 (2011)","DOI":"10.21437\/Interspeech.2011-169"},{"key":"8_CR103","doi-asserted-by":"crossref","unstructured":"Seltzer, M.L., Yu, D., Wang, Y.: An investigation of deep neural networks for noise robust speech recognition. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7398\u20137402. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"8_CR104","doi-asserted-by":"crossref","first-page":"101","DOI":"10.1016\/B978-0-08-051584-7.50012-7","volume-title":"Readings in Speech Recognition","author":"S. Seneff","year":"1990","unstructured":"Seneff, S.: A joint synchrony\/mean-rate model of auditory speech processing. In: Waibel, A., Lee, K.-F. (eds.) Readings in Speech Recognition, pp. 101\u2013111. Morgan Kaufmann, Burlington, MA (1990)"},{"issue":"1","key":"8_CR105","doi-asserted-by":"crossref","first-page":"77","DOI":"10.1016\/j.csl.2008.03.004","volume":"24","author":"Y. Shao","year":"2010","unstructured":"Shao, Y., Srinivasan, S., Jin, Z., Wang, D.: A computational auditory scene analysis system for speech segregation and robust speech recognition. Comput. Speech Lang. 24(1), 77\u201393 (2010)","journal-title":"Comput. Speech Lang."},{"issue":"7","key":"8_CR106","doi-asserted-by":"crossref","first-page":"2130","DOI":"10.1109\/TASL.2007.901836","volume":"15","author":"S. Srinivasan","year":"2007","unstructured":"Srinivasan, S., Wang, D.: Transforming binary uncertainties for robust speech recognition. IEEE Trans. Audio Speech Lang. Process. 15(7), 2130\u20132140 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"3","key":"8_CR107","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1121\/1.1915893","volume":"8","author":"S.S. Stevens","year":"1937","unstructured":"Stevens, S.S., Volkmann, J., Newman, E.B.: A scale for the measurement of the psychological magnitude pitch. J. Acoust. Soc. Am. 8(3), 185\u2013190 (1937)","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"8_CR108","doi-asserted-by":"crossref","first-page":"2040","DOI":"10.1121\/1.427950","volume":"106","author":"J. Tchorz","year":"1999","unstructured":"Tchorz, J., Kollmeier, B.: A model of auditory perception as front end for automatic speech recognition. J. Acoust. Soc. Am. 106(4), 2040\u20132050 (1999)","journal-title":"J. Acoust. Soc. Am."},{"issue":"5","key":"8_CR109","doi-asserted-by":"crossref","first-page":"599","DOI":"10.1109\/TASSP.1980.1163453","volume":"28","author":"H.M. Teager","year":"1980","unstructured":"Teager, H.M.: Some observations on oral air flow during phonation. IEEE Trans. Acoust. Speech Signal Process. 28(5), 599\u2013601 (1980)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"8_CR110","doi-asserted-by":"crossref","unstructured":"Thomas, S., Saon, G., Van Segbroeck, M., Narayanan, S.S.: Improvements to the IBM speech activity detection system for the DARPA RATS program. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4500\u20134504. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178822"},{"key":"8_CR111","doi-asserted-by":"crossref","unstructured":"T\u00fcske, Z., Golik, P., Schl\u00fcter, R., Ney, H.: Acoustic modeling with deep neural networks using raw time signal for LVCSR. In: Interspeech, pp. 890\u2013894 (2014)","DOI":"10.21437\/Interspeech.2014-223"},{"key":"8_CR112","volume-title":"Fepstrum features: design and application to conversational speech recognition","author":"V. Tyagi","year":"2011","unstructured":"Tyagi, V.: Fepstrum features: design and application to conversational speech recognition. Technical Report, IBM Research Report (2011)"},{"key":"8_CR113","volume-title":"Low complexity spectral imputation for noise robust speech recognition","author":"J. Hout Van","year":"2012","unstructured":"Van Hout, J.: Low complexity spectral imputation for noise robust speech recognition. M.S. thesis, UCLA (2012)"},{"issue":"5","key":"8_CR114","doi-asserted-by":"crossref","first-page":"1364","DOI":"10.1121\/1.383531","volume":"66","author":"N.F. Viemeister","year":"1979","unstructured":"Viemeister, N.F.: Temporal modulation transfer functions based upon modulation thresholds. J. Acoust. Soc. Am. 66(5), 1364\u20131380 (1979)","journal-title":"J. Acoust. Soc. Am."},{"issue":"2","key":"8_CR115","doi-asserted-by":"crossref","first-page":"126","DOI":"10.1109\/89.748118","volume":"7","author":"N. Virag","year":"1999","unstructured":"Virag, N.: Single channel speech enhancement based on masking properties of the human auditory system. IEEE Trans. Speech Audio Process. 7(2), 126\u2013137 (1999)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"8_CR116","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1007\/0-387-22794-6_12","volume-title":"Speech Separation by Humans and Machines","author":"D. Wang","year":"2005","unstructured":"Wang, D.: On ideal binary mask as the computational goal of auditory scene analysis. In: Divenyi, P. (ed.) Speech Separation by Humans and Machines, pp. 181\u2013197. Kluwer Academic, Dordrecht (2005)"},{"key":"8_CR117","volume-title":"The MERL\/MELCO\/TUM system for the REVERB challenge using deep recurrent neural network feature enhancement","author":"F. Weninger","year":"2014","unstructured":"Weninger, F., Watanabe, S., Le Roux, J., Hershey, J., Tachioka, Y., Geiger, J., Schuller, B., Rigoll, G.: The MERL\/MELCO\/TUM system for the REVERB challenge using deep recurrent neural network feature enhancement. In: Proceedings of the REVERB Workshop (2014)"},{"key":"8_CR118","doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Ragni, A., Gales, M.J.: Investigation of unsupervised adaptation of DNN acoustic models with filter bank input. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6344\u20136348. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6854825"},{"issue":"6","key":"8_CR119","doi-asserted-by":"crossref","first-page":"1896","DOI":"10.1121\/1.394754","volume":"81","author":"W.A. Yost","year":"1987","unstructured":"Yost, W.A., Moore, M.: Temporal changes in a complex spectral profile. J. Acoust. Soc. Am. 81(6), 1896\u20131905 (1987)","journal-title":"J. Acoust. Soc. Am."},{"key":"8_CR120","unstructured":"Yu, D., Seltzer, M.L., Li, J., Huang, J.T., Seide, F.: Feature learning in deep neural networks \u2013 studies on speech recognition tasks. arXiv:1301.3605 (2013, arXiv preprint)"},{"key":"8_CR121","doi-asserted-by":"crossref","unstructured":"Yu, D., Yao, K., Su, H., Li, G., Seide, F.: KL-divergence regularized deep neural network adaptation for improved large vocabulary speech recognition. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7893\u20137897. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639201"},{"key":"8_CR122","unstructured":"Zhan, P., Waibel, A.: Vocal tract length normalization for LVCSR. Technical Report, CMU-LTI-97-150, Carnegie Mellon University (1997)"},{"key":"8_CR123","volume-title":"Incorporating tandem\/HATS MLP features into SRIS conversational speech recognition system","author":"Q. Zhu","year":"2004","unstructured":"Zhu, Q., Stolcke, A., Chen, B.Y., Morgan, N.: Incorporating tandem\/HATS MLP features into SRIS conversational speech recognition system. In: Proceedings of the DARPA Rich Transcription Workshop (2004)"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T20:16:00Z","timestamp":1750968960000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":123,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_8","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}