{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T20:43:23Z","timestamp":1725914603943},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_19","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"401-417","source":"Crossref","is-referenced-by-count":0,"title":["Challenges in and Solutions to Deep Learning Network Acoustic Modeling in Speech Recognition Products at Microsoft"],"prefix":"10.1007","author":[{"given":"Yifan","family":"Gong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kshitiz","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinyu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaojun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoli","family":"Ye","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shixiong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"19_CR1","volume-title":"Automatic accent identification using Gaussian mixture models","author":"T. Chen","year":"2001","unstructured":"Chen, T., Huang, C., Chang, E., Wang, J.: Automatic accent identification using Gaussian mixture models. In: Proceedings of the Workshop on Automatic Speech Recognition and Understanding (2001)"},{"key":"19_CR2","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A., Hinton, G.: Speech recognition with deep recurrent neural networks. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a06645\u20136649 (2013)","DOI":"10.1109\/ICASSP.2013.6638947"},{"issue":"8","key":"19_CR3","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S. Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"19_CR4","volume-title":"Regularized sequence-level deep neural network model adaptation","author":"Y. Huang","year":"2015","unstructured":"Huang, Y., Gong, Y.: Regularized sequence-level deep neural network model adaptation. In: Proceedings of the Interspeech (2015)"},{"key":"19_CR5","doi-asserted-by":"crossref","unstructured":"Huang, J.T., Li, J., Yu, D., Deng, L., Gong, Y.: Cross-language knowledge transfer using multilingual deep neural network with shared hidden layers. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a07304\u20137308 (2013)","DOI":"10.1109\/ICASSP.2013.6639081"},{"key":"19_CR6","volume-title":"Semi-supervised GMM and DNN acoustic model training with multi-system combination and confidence re-calibration","author":"Y. Huang","year":"2013","unstructured":"Huang, Y., Yu, D., Gong, Y., Liu, C.: Semi-supervised GMM and DNN acoustic model training with multi-system combination and confidence re-calibration. In: Proceedings of the Interspeech (2013)"},{"key":"19_CR7","volume-title":"Multi-accent deep neural network acoustic model with accent-specific top layer using the KLD-regularized model adaptation","author":"Y. Huang","year":"2014","unstructured":"Huang, Y., Yu, D., Liu, C., Gong, Y.: Multi-accent deep neural network acoustic model with accent-specific top layer using the KLD-regularized model adaptation. In: Proceedings of the Interspeech (2014)"},{"key":"19_CR8","volume-title":"Semi-supervised training in deep learning acoustic models","author":"Y. Huang","year":"2016","unstructured":"Huang, Y., Wang, Y., Gong, Y.: Semi-supervised training in deep learning acoustic models. In: Proceedings of the Interspeech (2016)"},{"key":"19_CR9","volume-title":"Intermediate-layer DNN adaptation for offline and session-based iterative speaker adaptation","author":"K. Kumar","year":"2015","unstructured":"Kumar, K., Liu, C., Yao, K., Gong, Y.: Intermediate-layer DNN adaptation for offline and session-based iterative speaker adaptation. In: Sixteenth Annual Conference of the International Speech Communication Association (2015)"},{"key":"19_CR10","doi-asserted-by":"crossref","unstructured":"Kumar, K., Liu, C., Gong, Y.: Non-negative intermediate-layer DNN adaptation for a 10-kb speaker adaptation profile. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2016)","DOI":"10.1109\/ICASSP.2016.7472686"},{"key":"19_CR11","doi-asserted-by":"crossref","unstructured":"Li, J., Yu, D., Huang, J.T., Gong, Y.: Improving wideband speech recognition using mixed-bandwidth training data in CD-DNN-HMM. In: Proceedings of the IEEE Spoken Language Technology Workshop, pp.\u00a0131\u2013136 (2012)","DOI":"10.1109\/SLT.2012.6424210"},{"issue":"4","key":"19_CR12","doi-asserted-by":"crossref","first-page":"745","DOI":"10.1109\/TASLP.2014.2304637","volume":"22","author":"J. Li","year":"2014","unstructured":"Li, J., Deng, L., Gong, Y., Haeb-Umbach, R.: An overview of noise-robust automatic speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(4), 745\u2013777 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"19_CR13","volume-title":"Factorized adaptation for deep neural network","author":"J. Li","year":"2014","unstructured":"Li, J., Huang, J.T., Gong, Y.: Factorized adaptation for deep neural network. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (2014)"},{"key":"19_CR14","volume-title":"Learning small-size DNN with output-distribution-based criteria","author":"J. Li","year":"2014","unstructured":"Li, J., Zhao, R., Huang, J.T., Gong, Y.: Learning small-size DNN with output-distribution-based criteria. In: Proceedings of the Interspeech (2014)"},{"key":"19_CR15","volume-title":"Robust Automatic Speech Recognition: A Bridge to Practical Applications","author":"J. Li","year":"2015","unstructured":"Li, J., Deng, L., Haeb-Umbach, R., Gong, Y.: Robust Automatic Speech Recognition: A Bridge to Practical Applications. Academic, London (2015)"},{"key":"19_CR16","volume-title":"LSTM time and frequency recurrence for automatic speech recognition","author":"J. Li","year":"2015","unstructured":"Li, J., Mohamed, A., Zweig, G., Gong, Y.: LSTM time and frequency recurrence for automatic speech recognition. In: Proceedings of the Workshop on Automatic Speech Recognition and Understanding (2015)"},{"key":"19_CR17","volume-title":"Exploring multidimensional LSTMs for large vocabulary ASR","author":"J. Li","year":"2016","unstructured":"Li, J., Mohamed, A., Zweig, G., Gong, Y.: Exploring multidimensional LSTMs for large vocabulary ASR. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (2016)"},{"key":"19_CR18","volume-title":"Simplifying long short-term memory acoustic models for fast training and decoding","author":"Y. Miao","year":"2016","unstructured":"Miao, Y., Li, J., Wang, Y., Zhang, S., Gong, Y.: Simplifying long short-term memory acoustic models for fast training and decoding. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (2016)"},{"key":"19_CR19","doi-asserted-by":"crossref","unstructured":"Mohamed, A., Hinton, G., Penn, G.: Understanding how deep belief networks perform acoustic modelling. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a04273\u20134276 (2012)","DOI":"10.1109\/ICASSP.2012.6288863"},{"key":"19_CR20","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A., Beaufays, F.: Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In: INTERSPEECH, pp.\u00a0338\u2013342 (2014)","DOI":"10.21437\/Interspeech.2014-80"},{"key":"19_CR21","volume-title":"Error back propagation for sequence training of context-dependent deep networks for conversational speech transcription","author":"H. Su","year":"2013","unstructured":"Su, H., Li, G., Yu, D., Seide, F.: Error back propagation for sequence training of context-dependent deep networks for conversational speech transcription. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (2013)"},{"key":"19_CR22","doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Renals, S.: Learning hidden unit contributions for unsupervised speaker adaptation of neural network acoustic models. In: Proceedings of the SLT, pp.\u00a0171\u2013176 (2014)","DOI":"10.1109\/SLT.2014.7078569"},{"key":"19_CR23","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4757-2440-0","volume-title":"The Nature of Statistical Learning Theory","author":"V.N. Vapnik","year":"1995","unstructured":"Vapnik, V.N.: The Nature of Statistical Learning Theory. Springer, New York (1995)"},{"key":"19_CR24","doi-asserted-by":"crossref","unstructured":"Xue, J., Li, J., Gong, Y.: Restructuring of deep neural network acoustic models with singular value decomposition. In: Proceedings of the Interspeech, pp.\u00a02365\u20132369 (2013)","DOI":"10.21437\/Interspeech.2013-552"},{"key":"19_CR25","doi-asserted-by":"crossref","unstructured":"Xue, J., Li, J., Yu, D., Seltzer, M., Gong, Y.: Singular value decomposition based low-footprint speaker adaptation and personalization for deep neural network. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a06359\u20136363 (2014)","DOI":"10.1109\/ICASSP.2014.6854828"},{"key":"19_CR26","doi-asserted-by":"crossref","unstructured":"Ye, G., Liu, C., Gong, Y.: Geo-location dependent deep neural network acoustic model for speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a05870\u20135874 (2016)","DOI":"10.1109\/ICASSP.2016.7472803"},{"key":"19_CR27","volume-title":"KL-divergence regularized deep neural network adaptation for improved large vocabulary speech recognition","author":"D. Yu","year":"2013","unstructured":"Yu, D., Yao, K., Su, H., Li, G., Seide, F.: KL-divergence regularized deep neural network adaptation for improved large vocabulary speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (2013)"},{"key":"19_CR28","doi-asserted-by":"crossref","unstructured":"Zhang, S.X., Liu, C., Yao, K., Gong, Y.: Deep neural support vector machines for speech recognition. In: ICASSP, pp.\u00a04275\u20134279. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178777"},{"key":"19_CR29","volume-title":"Recurrent support vector machines for speech recognition","author":"S.X. Zhang","year":"2016","unstructured":"Zhang, S.X., Zhao, R., Liu, C., Li, J., Gong, Y.: Recurrent support vector machines for speech recognition. In: ICASSP. IEEE, New York (2016)"},{"key":"19_CR30","volume-title":"Variable-activation and variable-input deep neural network for robust speech recognition","author":"R. Zhao","year":"2014","unstructured":"Zhao, R., Li, J., Gong, Y.: Variable-activation and variable-input deep neural network for robust speech recognition. In: Proceedings of the IEEE Spoken Language Technology Workshop (2014)"},{"key":"19_CR31","volume-title":"Variable-component deep neural network for robust speech recognition","author":"R. Zhao","year":"2014","unstructured":"Zhao, R., Li, J., Gong, Y.: Variable-component deep neural network for robust speech recognition. In: Proceedings of the Interspeech (2014)"},{"key":"19_CR32","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Li, J., Xue, J., Gong, Y.: Investigating online low-footprint speaker adaptation using generalized linear regression and click-through data. In: Proceedings of the ICASSP, pp.\u00a04310\u20134314 (2015)","DOI":"10.1109\/ICASSP.2015.7178784"},{"key":"19_CR33","volume-title":"Low-rank plus diagonal adaptation for deep neural networks","author":"Y. Zhao","year":"2016","unstructured":"Zhao, Y., Li, J., Gong, Y.: Low-rank plus diagonal adaptation for deep neural networks. In: Proceedings of the ICASSP (2016)"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,27]],"date-time":"2023-08-27T19:50:10Z","timestamp":1693165810000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_19","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}