{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T22:06:32Z","timestamp":1773093992635,"version":"3.50.1"},"publisher-location":"Cham","reference-count":71,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319646794","type":"print"},{"value":"9783319646800","type":"electronic"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_13","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T04:37:26Z","timestamp":1509424646000},"page":"299-323","source":"Crossref","is-referenced-by-count":3,"title":["End-to-End Architectures for Speech Recognition"],"prefix":"10.1007","author":[{"given":"Yajie","family":"Miao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Florian","family":"Metze","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1007\/978-3-540-76336-9_3","volume-title":"Implementation and Application of Automata","author":"C. Allauzen","year":"2007","unstructured":"Allauzen, C., Riley, M., Schalkwyk, J., Skut, W., Mohri, M.: OpenFST: a general and efficient weighted finite-state transducer library. In: Holub, J., \u017dd\u00e1vn, J. (eds.) Implementation and Application of Automata, pp.\u00a011\u201323. Springer, Heidelberg (2007)"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Bacchiani, M., Senior, A., Heigold, G.: Asynchronous, online, GMM-free training of a context dependent acoustic model for speech recognition. In: Fifteenth Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Singapore (2014)","DOI":"10.21437\/Interspeech.2014-430"},{"key":"13_CR3","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate (2014). arXiv preprint arXiv:1409.0473"},{"key":"13_CR4","unstructured":"Bahdanau, D., Serdyuk, D., Brakel, P., Ke, N.R., Chorowski, J., Courville, A.C., Bengio, Y.: Task loss estimation for sequence prediction. CoRR abs\/1511.06456 (2015). http:\/\/arxiv.org\/abs\/1511.06456"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Bahdanau, D., Chorowski, J., Serdyuk, D., Brakel, P., Bengio, Y.: End-to-end attention-based large vocabulary speech recognition. In: Seventeenth Annual Conference of the International Speech Communication Association (INTERSPEECH) (2016)","DOI":"10.1109\/ICASSP.2016.7472618"},{"issue":"2","key":"13_CR6","doi-asserted-by":"crossref","first-page":"179","DOI":"10.1109\/TPAMI.1983.4767370","volume":"5","author":"L.R. Bahl","year":"1983","unstructured":"Bahl, L.R., Jelinek, F., Mercer, R.L.: A maximum likelihood approach to continuous speech recognition. IEEE Trans. Pattern Anal. Mach. Intell. 5(2), 179\u2013190 (1983)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"2","key":"13_CR7","doi-asserted-by":"crossref","first-page":"157","DOI":"10.1109\/72.279181","volume":"5","author":"Y. Bengio","year":"1994","unstructured":"Bengio, Y., Simard, P., Frasconi, P.: Learning long-term dependencies with gradient descent is difficult. IEEE Trans. Neural Netw. 5(2), 157\u2013166 (1994)","journal-title":"IEEE Trans. Neural Netw."},{"key":"13_CR8","first-page":"1137","volume":"3","author":"Y. Bengio","year":"2003","unstructured":"Bengio, Y., Ducharme, R., Vincent, P., Jauvin, C.: A neural probabilistic language model. J. Mach. Learn. Res. 3, 1137\u20131155 (2003)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR9","unstructured":"Chan, W., Jaitly, N., Le, Q.V., Vinyals, O.: Listen, attend and spell (2015). arXiv preprint arXiv:1508.01211"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Chan, W., Jaitly, N., Le, Q.V., Vinyals, O.: Listen, attend and spell: a neural network for large vocabulary conversational speech recognition. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, New York (2016)","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"13_CR11","unstructured":"Cho, K., van Merrienboer, B., Bahdanau, D., Bengio, Y.: On the properties of neural machine translation: encoder\u2013decoder approaches. CoRR abs\/1409.1259 (2014). http:\/\/arxiv.org\/abs\/1409.1259"},{"key":"13_CR12","unstructured":"Cho, K., Van\u00a0Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using RNN encoder\u2013decoder for statistical machine translation (2014). arXiv preprint arXiv:1406.1078"},{"key":"13_CR13","unstructured":"Chorowski, J., Bahdanau, D., Cho, K., Bengio, Y.: End-to-end continuous speech recognition using attention-based recurrent NN: first results (2014). arXiv preprint arXiv:1412.1602"},{"key":"13_CR14","unstructured":"Chorowski, J.K., Bahdanau, D., Serdyuk, D., Cho, K., Bengio, Y.: Attention-based models for speech recognition. In: Advances in Neural Information Processing Systems, pp.\u00a0577\u2013585 (2015)"},{"key":"13_CR15","unstructured":"Collobert, R., Puhrsch, C., Synnaeve, G.: Wav2Letter: an end-to-end convNet-based speech recognition system. CoRR abs\/1609.03193 (2016). http:\/\/arxiv.org\/abs\/1609.03193"},{"issue":"1","key":"13_CR16","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G.E. Dahl","year":"2012","unstructured":"Dahl, G.E., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(1), 30\u201342 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Fern\u00e1ndez, S., Graves, A., Schmidhuber, J.: An application of recurrent neural networks to discriminative keyword spotting. In: Artificial Neural Networks\u2013ICANN 2007, pp.\u00a0220\u2013229. Springer, Heidelberg (2007)","DOI":"10.1007\/978-3-540-74695-9_23"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Garofolo, J.S., Lamel, L.F., Fisher, W.M., Fiscus, J.G., Pallett, D.S.: DARPA TIMIT acoustic-phonetic continuous speech corpus CD-ROM. NIST speech disc 1-1.1. NASA STI\/Recon Technical Report N 93 (1993)","DOI":"10.6028\/NIST.IR.4930"},{"key":"13_CR19","unstructured":"Geras, K.J., Mohamed, A.R., Caruana, R., Urban, G., Wang, S., Aslan, O., Philipose, M., Richardson, M., Sutton, C.: Blending LSTMS into CNNS (2015). arXiv preprint arXiv:1511.06433"},{"key":"13_CR20","first-page":"115","volume":"3","author":"F.A. Gers","year":"2003","unstructured":"Gers, F.A., Schraudolph, N.N., Schmidhuber, J.: Learning precise timing with LSTM recurrent networks. J. Mach. Learn. Res. 3, 115\u2013143 (2003)","journal-title":"J. Mach. Learn. Res."},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"Ghahremani, P., BabaAli, B., Povey, D., et\u00a0al.: A pitch extraction algorithm tuned for automatic speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a02494\u20132498. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854049"},{"key":"13_CR22","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a0580\u2013587 (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Godfrey, J.J., Holliman, E.C., McDaniel, J.: Switchboard: telephone speech corpus for research and development. In: 1992 IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP-92, vol.\u00a01, pp.\u00a0517\u2013520. IEEE, New York (1992)","DOI":"10.1109\/ICASSP.1992.225858"},{"key":"13_CR24","unstructured":"Graves, A.: Sequence transduction with recurrent neural networks (2012). arXiv preprint arXiv:1211.3711"},{"key":"13_CR25","unstructured":"Graves, A., Jaitly, N.: Towards end-to-end speech recognition with recurrent neural networks. In: Proceedings of the 31st International Conference on Machine Learning (ICML-14), pp.\u00a01764\u20131772 (2014)"},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning (ICML-06), pp.\u00a0369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"13_CR27","unstructured":"Hannun, A., Case, C., Casper, J., Catanzaro, B., Diamos, G., Elsen, E., Prenger, R., Satheesh, S., Sengupta, S., Coates, A., et\u00a0al.: Deepspeech: scaling up end-to-end speech recognition (2014). arXiv preprint arXiv:1412.5567"},{"key":"13_CR28","unstructured":"Hannun, A.Y., Maas, A.L., Jurafsky, D., Ng, A.Y.: First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv preprint arXiv:1408.2873 (2014)"},{"key":"13_CR29","doi-asserted-by":"crossref","first-page":"599","DOI":"10.1007\/978-3-642-35289-8_32","volume-title":"Neural Networks: Tricks of the Trade","author":"G.E. Hinton","year":"2012","unstructured":"Hinton, G.E.: A practical guide to training restricted Boltzmann machines. In: Montavon, G., Orr, G., M\u00fcller, K.R. (eds.) Neural Networks: Tricks of the Trade, pp.\u00a0599\u2013619. Springer, Heidelberg (2012)"},{"issue":"8","key":"13_CR30","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S. Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Hoshen, Y., Weiss, R.J., Wilson, K.W.: Speech acoustic modeling from raw multichannel waveforms. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a04624\u20134628. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178847"},{"key":"13_CR32","unstructured":"Hwang, K., Sung, W.: Online sequence training of recurrent neural networks with connectionist temporal classification (2015). arXiv preprint arXiv:1511.06841"},{"key":"13_CR33","unstructured":"Kalchbrenner, N., Blunsom, P.: Recurrent convolutional neural networks for discourse compositionality. CoRR abs\/1306.3584 (2013). http:\/\/arxiv.org\/abs\/1306.3584"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Toderici, G., Shetty, S., Leung, T., Sukthankar, R., Fei-Fei, L.: Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a01725\u20131732 (2014)","DOI":"10.1109\/CVPR.2014.223"},{"key":"13_CR35","unstructured":"Karpathy, A., Johnson, J., Li, F.F.: Visualizing and understanding recurrent networks (2015). arXiv preprint arXiv:1506.02078"},{"key":"13_CR36","unstructured":"Kilgour, K.: Modularity and neural integration in large-vocabulary continuous speech recognition. Ph.D. thesis, Karlsruhe Institute of Technology (2015)"},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Kingsbury, B.: Lattice-based optimization of sequence classification criteria for neural-network acoustic modeling. In: 2009 IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a03761\u20133764. IEEE, New York (2009)","DOI":"10.1109\/ICASSP.2009.4960445"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"Koehn, P., Och, F.J., Marcu, D.: Statistical phrase-based translation. In: Proceedings of the 2003 Conference of the North American Chapter of the Association for Computational Linguistics on Human Language Technology, vol.\u00a01, pp.\u00a048\u201354. Association for Computational Linguistics, Stroudsburg (2003)","DOI":"10.21236\/ADA461156"},{"key":"13_CR39","doi-asserted-by":"crossref","unstructured":"Li, H., Lin, Z., Shen, X., Brandt, J., Hua, G.: A convolutional neural network cascade for face detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a05325\u20135334 (2015)","DOI":"10.1109\/CVPR.2015.7299170"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, H., Cai, X., Xu, B.: Towards end-to-end speech recognition for Chinese Mandarin using long short-term memory recurrent neural networks. In: Sixteenth Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Dresden (2015)","DOI":"10.21437\/Interspeech.2015-717"},{"key":"13_CR41","doi-asserted-by":"crossref","unstructured":"Liu, Y., Fung, P., Yang, Y., Cieri, C., Huang, S., Graff, D.: HKUST\/MTS: a very large scale Mandarin telephone speech corpus. In: Chinese Spoken Language Processing, pp.\u00a0724\u2013735 (2006)","DOI":"10.1007\/11939993_73"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Lowe, D.G.: Object recognition from local scale-invariant features. In: Proceedings of the Seventh IEEE International Conference on Computer Vision, 1999, vol.\u00a02, pp.\u00a01150\u20131157. IEEE, New York (1999)","DOI":"10.1109\/ICCV.1999.790410"},{"key":"13_CR43","volume-title":"A study of the recurrent neural network encoder-decoder for large vocabulary speech recognition","author":"L. Lu","year":"2015","unstructured":"Lu, L., Zhang, X., Cho, K., Renals, S.: A study of the recurrent neural network encoder-decoder for large vocabulary speech recognition. In: Sixteenth Annual Conference of the International Speech Communication Association (2015)"},{"key":"13_CR44","unstructured":"Lu, L., Kong, L., Dyer, C., Smith, N.A., Renals, S.: Segmental recurrent neural networks for end-to-end speech recognition. CoRR abs\/1603.00223 (2016). http:\/\/arxiv.org\/abs\/1603.00223"},{"key":"13_CR45","doi-asserted-by":"crossref","unstructured":"Lu, L., Zhang, X., Renals, S.: On training the recurrent neural network encoder\u2013decoder for large vocabulary end-to-end speech recognition. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, New York (2016)","DOI":"10.1109\/ICASSP.2016.7472641"},{"key":"13_CR46","doi-asserted-by":"crossref","unstructured":"Maas, A.L., Xie, Z., Jurafsky, D., Ng, A.Y.: Lexicon-free conversational speech recognition with neural networks. In: Proceedings of the 2015 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (2015)","DOI":"10.3115\/v1\/N15-1038"},{"issue":"4","key":"13_CR47","doi-asserted-by":"crossref","first-page":"373","DOI":"10.1006\/csla.2000.0152","volume":"14","author":"L. Mangu","year":"2000","unstructured":"Mangu, L., Brill, E., Stolcke, A.: Finding consensus in speech recognition: word error minimization and other applications of confusion networks. Comput. Speech Lang. 14(4), 373\u2013400 (2000)","journal-title":"Comput. Speech Lang."},{"key":"13_CR48","volume-title":"The speech recognition virtual kitchen","author":"F. Metze","year":"2013","unstructured":"Metze, F., Fosler-Lussier, E., Bates, R.: The speech recognition virtual kitchen. In: Proceedings of INTERSPEECH. ISCA, Lyon, France (2013). https:\/\/github.com\/srvk\/eesen-transcriber"},{"key":"13_CR49","doi-asserted-by":"crossref","unstructured":"Miao, Y., Metze, F.: On speaker adaptation of long short-term memory recurrent neural networks. In: Sixteenth Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Dresden (2015)","DOI":"10.21437\/Interspeech.2015-290"},{"key":"13_CR50","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Metze, F.: EESEN: end-to-end speech recognition using deep RNN models and WFST-based decoding. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU). IEEE, New York (2015)","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"13_CR51","doi-asserted-by":"crossref","unstructured":"Mohamed, A.R., Seide, F., Yu, D., Droppo, J., Stoicke, A., Zweig, G., Penn, G.: Deep bi-directional recurrent networks over spectral windows. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a078\u201383. IEEE, New York (2015)","DOI":"10.1109\/ASRU.2015.7404777"},{"issue":"1","key":"13_CR52","doi-asserted-by":"crossref","first-page":"69","DOI":"10.1006\/csla.2001.0184","volume":"16","author":"M. Mohri","year":"2002","unstructured":"Mohri, M., Pereira, F., Riley, M.: Weighted finite-state transducers in speech recognition. Comput. Speech Lang. 16(1), 69\u201388 (2002)","journal-title":"Comput. Speech Lang."},{"key":"13_CR53","unstructured":"Nair, V., Hinton, G.E.: Rectified linear units improve restricted Boltzmann machines. In: Proceedings of the 27th International Conference on Machine Learning (ICML-10), pp.\u00a0807\u2013814 (2010)"},{"key":"13_CR54","unstructured":"Palaz, D., Collobert, R., Doss, M.M.: Estimating phoneme class conditional probabilities from raw speech signal using convolutional neural networks (2013). arXiv preprint arXiv:1304.1018"},{"key":"13_CR55","doi-asserted-by":"crossref","unstructured":"Paul, D.B., Baker, J.M.: The design for the wall street journal-based CSR corpus. In: Proceedings of the Workshop on Speech and Natural Language, pp.\u00a0357\u2013362. Association for Computational Linguistics, Morristown (1992)","DOI":"10.3115\/1075527.1075614"},{"key":"13_CR56","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motl\u00ed\u010dek, P., Qian, Y., Schwarz, P., Silovsk\u00fd, J., Stemmer, G., Vesel\u00fd, K.: The Kaldi speech recognition toolkit. In: 2011 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 1\u20134. IEEE, New York (2011)"},{"issue":"2","key":"13_CR57","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1109\/5.18626","volume":"77","author":"L.R. Rabiner","year":"1989","unstructured":"Rabiner, L.R.: A tutorial on hidden Markov models and selected applications in speech recognition. Proc. IEEE 77(2), 257\u2013286 (1989)","journal-title":"Proc. IEEE"},{"key":"13_CR58","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Vinyals, O., Senior, A., Sak, H.: Convolutional, long short-term memory, fully connected deep neural networks. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a04580\u20134584. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"13_CR59","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Weiss, R.J., Senior, A., Wilson, K.W., Vinyals, O.: Learning the speech front-end with raw waveform CLDNNs. In: Sixteenth Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Dresden (2015)","DOI":"10.21437\/Interspeech.2015-1"},{"key":"13_CR60","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A., Beaufays, F.: Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In: Fifteenth Annual Conference of the International Speech Communication Association (INTERSPEECH). ISCA, Singapore (2014)","DOI":"10.21437\/Interspeech.2014-80"},{"key":"13_CR61","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A., Rao, K., Irsoy, O., Graves, A., Beaufays, F., Schalkwyk, J.: Learning acoustic frame labeling for speech recognition with recurrent neural networks. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a04280\u20134284. IEEE, New York (2015)","DOI":"10.1109\/ICASSP.2015.7178778"},{"key":"13_CR62","doi-asserted-by":"crossref","unstructured":"Senior, A., Heigold, G., Bacchiani, M., Liao, H.: GMM-free DNN training. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a05639\u20135643. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854675"},{"key":"13_CR63","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. In: Advances in Neural Information Processing Systems, pp.\u00a03104\u20133112 (2014)"},{"key":"13_CR64","doi-asserted-by":"crossref","unstructured":"T\u00fcske, Z., Golik, P., Schl\u00fcter, R., Ney, H.: Acoustic modeling with deep neural networks using raw time signal for LVCSR. In: Fifteenth Annual Conference of the International Speech Communication Association (INTERSPEECH), pp.\u00a0890\u2013894. ISCA, Singapore (2014)","DOI":"10.21437\/Interspeech.2014-223"},{"key":"13_CR65","doi-asserted-by":"crossref","unstructured":"Watanabe, S., Le\u00a0Roux, J.: Black box optimization for automatic speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a03256\u20133260. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854202"},{"key":"13_CR66","doi-asserted-by":"crossref","unstructured":"W\u00f6llmer, M., Eyben, F., Schuller, B., Rigoll, G.: Spoken term detection with connectionist temporal classification: a novel hybrid CTC-DBN decoder. In: 2010 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a05274\u20135277. IEEE, New York (2010)","DOI":"10.1109\/ICASSP.2010.5494980"},{"key":"13_CR67","unstructured":"Xu, K., Ba, J., Kiros, R., Courville, A., Salakhutdinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: neural image caption generation with visual attention (2015). arXiv preprint arXiv:1502.03044"},{"key":"13_CR68","doi-asserted-by":"crossref","unstructured":"Yao, L., Torabi, A., Cho, K., Ballas, N., Pal, C., Larochelle, H., Courville, A.: Describing videos by exploiting temporal structure. In: Proceedings of the IEEE International Conference on Computer Vision, pp.\u00a04507\u20134515 (2015)","DOI":"10.1109\/ICCV.2015.512"},{"key":"13_CR69","doi-asserted-by":"crossref","unstructured":"Zhang, C., Woodland, P.C.: Standalone training of context-dependent deep neural network acoustic models. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a05597\u20135601. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854674"},{"key":"13_CR70","unstructured":"Zhou, B., Lapedriza, A., Xiao, J., Torralba, A., Oliva, A.: Learning deep features for scene recognition using places database. In: Advances in Neural Information Processing Systems, pp.\u00a0487\u2013495 (2014)"},{"key":"13_CR71","unstructured":"Zweig, G., Yu, C., Droppo, J., Stolcke, A.: Advances in all-neural speech recognition (2016). arXiv:1609.05935"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,27]],"date-time":"2023-08-27T15:50:08Z","timestamp":1693151408000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":71,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_13","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}