{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T16:46:39Z","timestamp":1762015599187},"publisher-location":"Cham","reference-count":35,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_11","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"261-279","source":"Crossref","is-referenced-by-count":1,"title":["Advanced Recurrent Neural Networks for Automatic Speech Recognition"],"prefix":"10.1007","author":[{"given":"Yu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guoguo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"11_CR1","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O. Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid, O., Mohamed, A., Jiang, H., Deng, L., Penn, G., Yu, D.: Convolutional neural networks for speech recognition. IEEE Trans. Audio Speech Lang. Process. 22, 1533\u20131545 (2014). doi:10.1109\/TASLP.2014.2339736","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"11_CR2","volume-title":"Regularization of context-dependent deep neural networks with context-independent multi-task training","author":"P. Bell","year":"2015","unstructured":"Bell, P., Renals, S.: Regularization of context-dependent deep neural networks with context-independent multi-task training. In: Proceedings of ICASSP (2015)"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Bi, M., Qian, Y., Yu, K.: Very deep convolutional neural networks for LVCSR. In: Proceedings of Annual Conference of International Speech Communication Association (INTERSPEECH) (2015)","DOI":"10.21437\/Interspeech.2015-656"},{"issue":"2","key":"11_CR4","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1007\/s10579-007-9040-x","volume":"41","author":"J. Carletta","year":"2007","unstructured":"Carletta, J.: Unleashing the killer corpus: experiences in creating the multi-everything AMI meeting corpus. Lang. Resour. Eval. J. 41(2), 181\u2013190 (2007)","journal-title":"Lang. Resour. Eval. J."},{"key":"11_CR5","volume-title":"Training deep bidirectional LSTM acoustic model for LVCSR by a context-sensitive-chunk BPTT approach","author":"K. Chen","year":"2015","unstructured":"Chen, K., Yan, Z.J., Huo, Q.: Training deep bidirectional LSTM acoustic model for LVCSR by a context-sensitive-chunk BPTT approach. In: INTERSPEECH (2015)"},{"key":"11_CR6","volume-title":"Multilingual data selection for training stacked bottleneck features","author":"E. Chuangsuwanich","year":"2016","unstructured":"Chuangsuwanich, E., Zhang, Y., Glass, J.: Multilingual data selection for training stacked bottleneck features. In: Proceedings of ICASSP (2016)"},{"issue":"1","key":"11_CR7","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G.E. Dahl","year":"2012","unstructured":"Dahl, G.E., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(1), 30\u201342 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"11_CR8","volume-title":"Multiple dimension Levenshtein edit distance calculations for evaluating ASR systems during simultaneous speech","author":"J. Fiscus","year":"2006","unstructured":"Fiscus, J., Ajot, J., Radde, N., Laprun, C.: Multiple dimension Levenshtein edit distance calculations for evaluating ASR systems during simultaneous speech. In: LREC (2006)"},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Ghahremani, P., Droppo, J., Seltzer, M.L.: Linearly augmented deep neural network. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2016)","DOI":"10.1109\/ICASSP.2016.7472646"},{"key":"11_CR10","doi-asserted-by":"crossref","unstructured":"Graves, A., Jaitly, N., Mohamed, A.: Hybrid speech recognition with deep bidirectional LSTM. In: Proceedings of IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a0273\u2013278 (2013)","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A., Hinton, G.: Speech recognition with deep recurrent neural networks. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2013)","DOI":"10.1109\/ICASSP.2013.6638947"},{"issue":"2","key":"11_CR12","doi-asserted-by":"crossref","first-page":"486","DOI":"10.1109\/TASL.2011.2163395","volume":"20","author":"T. Hain","year":"2012","unstructured":"Hain, T., Burget, L., Dines, J., Garner, P.N., Grzl, F., Hannani, A.E., Huijbregts, M., Karafit, M., Lincoln, M., Wan, V.: Transcribing meetings with the AMIDA systems. IEEE Trans. Audio Speech Lang. Process. 20(2), 486\u2013498 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"11_CR13","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. CoRR abs\/1512.03385 (2015). http:\/\/arxiv.org\/abs\/1512.03385"},{"key":"11_CR14","volume-title":"Asynchronous stochastic optimization for sequence training of deep neural networks","author":"G. Heigold","year":"2014","unstructured":"Heigold, G., McDermott, E., Vanhoucke, V., Senior, A., Bacchiani, M.: Asynchronous stochastic optimization for sequence training of deep neural networks. In: ICASSP (2014)"},{"issue":"6","key":"11_CR15","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G. Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G., Mohamed, A., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T., Kingsbury, B.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"11_CR16","unstructured":"Hinton, G.E., Srivastava, N., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.: Improving neural networks by preventing co-adaptation of feature detectors. http:\/\/arxiv.org\/abs\/1207.0580 (2012)"},{"issue":"8","key":"11_CR17","doi-asserted-by":"crossref","first-page":"17351438","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S. Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 17351438 (1997)","journal-title":"Neural Comput."},{"key":"11_CR18","volume-title":"Babel program broad agency announcement, IARPA-BAA-11-02","author":"IARPA","year":"2011","unstructured":"IARPA: Babel program broad agency announcement, IARPA-BAA-11-02 (2011)"},{"key":"11_CR19","unstructured":"Kalchbrenner, N., Danihelka, I., Graves, A.: Grid long short-term memory. http:\/\/arXiv.org\/abs\/1507.01526 (2015)"},{"issue":"6","key":"11_CR20","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1109\/MSP.2012.2205285","volume":"29","author":"K. Kumatani","year":"2012","unstructured":"Kumatani, K., McDonough, J.W., Raj, B.: Microphone array processing for distant speech recognition: from close-talking microphones to far-field sensors. IEEE Signal Process. Mag. 29(6), 127\u2013140 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"11_CR21","volume-title":"The Kaldi speech recognition toolkit","author":"D. Povey","year":"2011","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motl\u00ed\u010dek, P., Qian, Y., Schwarz, P., Silovsk\u00fd, J., Stemmer, G., Vesel\u00fd, K.: The Kaldi speech recognition toolkit. In: ASRU (2011)"},{"key":"11_CR22","volume-title":"Long short-term memory recurrent neural network architectures for large scale acoustic modeling","author":"H. Sak","year":"2014","unstructured":"Sak, H., Senior, A., Beaufays, F.: Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In: Fifteenth Annual Conference of the International Speech Communication Association (2014)"},{"key":"11_CR23","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: Proceedings of IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a024\u201329 (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Yu, D.: Conversational speech transcription using context-dependent deep neural networks. In: Proceedings of Annual Conference of International Speech Communication Association (INTERSPEECH), pp.\u00a0437\u2013440 (2011)","DOI":"10.21437\/Interspeech.2011-169"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Seltzer, M., Yu, D., Wang, Y.Q.: An investigation of deep neural networks for noise robust speech recognition. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2013)","DOI":"10.1109\/ICASSP.2013.6639100"},{"key":"11_CR26","volume-title":"Making the most from multiple microphones in meeting recognition","author":"A. Stolcke","year":"2011","unstructured":"Stolcke, A.: Making the most from multiple microphones in meeting recognition. In: ICASSP (2011)"},{"key":"11_CR27","volume-title":"Hybrid acoustic models for distant and multichannel large vocabulary speech recognition","author":"P. Swietojanski","year":"2013","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Hybrid acoustic models for distant and multichannel large vocabulary speech recognition. In: ASRU (2013)"},{"issue":"9","key":"11_CR28","doi-asserted-by":"publisher","first-page":"1120","DOI":"10.1109\/LSP.2014.2325781","volume":"21","author":"P. Swietojanski","year":"2014","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Convolutional neural networks for distant speech recognition. IEEE Signal Process. Lett. 21(9), 1120\u20131124 (2014). doi:10.1109\/LSP.2014.2325781","journal-title":"IEEE Signal Process. Lett."},{"key":"11_CR29","volume-title":"Structured output layer with auxiliary targets for context-dependent acoustic modelling","author":"P. Swietojanski","year":"2015","unstructured":"Swietojanski, P., Bell, P., Renals, S.: Structured output layer with auxiliary targets for context-dependent acoustic modelling. In: Proceedings of INTERSPEECH (2015)"},{"key":"11_CR30","doi-asserted-by":"crossref","first-page":"490501","DOI":"10.1162\/neco.1990.2.4.490","volume":"2","author":"R. Williams","year":"1990","unstructured":"Williams, R., Peng, J.: An efficient gradient-based algorithm for online training of recurrent network trajectories. Neural Comput. 2, 490501 (1990)","journal-title":"Neural Comput."},{"key":"11_CR31","unstructured":"Yu, D., Eversole, A., Seltzer, M., Yao, K., Guenter, B., Kuchaiev, O., Zhang, Y., Seide, F., Chen, G., Wang, H., Droppo, J., Agarwal, A., Basoglu, C., Padmilac, M., Kamenev, A., Ivanov, V., Cyphers, S., Parthasarathi, H., Mitra, B., Huang, Z., Zweig, G., Rossbach, C., Currey, J., Gao, J., May, A., Peng, B., Stolcke, A., Slaney, M., Huang, X.: An introduction to computational networks and the computational network toolkit. Microsoft Technical Report (2014)"},{"key":"11_CR32","doi-asserted-by":"crossref","unstructured":"Yu, D., Xiong, W., Droppo, J., Stolcke, A., Ye, G., Li, J., Zweig, G.: Deep convolutional neural networks with layer-wise context expansion and attention. In: Proceedings of Annual Conference of International Speech Communication Association (INTERSPEECH) (2016)","DOI":"10.21437\/Interspeech.2016-251"},{"key":"11_CR33","volume-title":"Speech recognition with prediction\u2013adaptation\u2013correction recurrent neural networks","author":"Y. Zhang","year":"2015","unstructured":"Zhang, Y., Yu, D., Seltzer, M., Droppo, J.: Speech recognition with prediction\u2013adaptation\u2013correction recurrent neural networks. In: Proceedings of ICASSP (2015)"},{"key":"11_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, G., Yu, D., Yao, K., Khudanpur, S., Glass, J.: Highway long short-term memory RNNS for distant speech recognition. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2016)","DOI":"10.1109\/ICASSP.2016.7472780"},{"key":"11_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chuangsuwanich, E., Glass, J., Yu, D.: Prediction\u2013adaptation\u2013correction recurrent neural networks for low-resource language speech recognition. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP) (2016)","DOI":"10.1109\/ICASSP.2016.7472712"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T22:20:14Z","timestamp":1659738014000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_11","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}