{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T12:17:48Z","timestamp":1767183468241},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319592589"},{"type":"electronic","value":"9783319592596"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-59259-6_9","type":"book-chapter","created":{"date-parts":[[2017,5,31]],"date-time":"2017-05-31T02:56:53Z","timestamp":1496199413000},"page":"98-109","source":"Crossref","is-referenced-by-count":10,"title":["Audio Visual Speech Recognition Using Deep Recurrent Neural Networks"],"prefix":"10.1007","author":[{"given":"Abhinav","family":"Thanda","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shankar M.","family":"Venkatesan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,6,1]]},"reference":[{"key":"9_CR1","unstructured":"Assael, Y.M., Shillingford, B., Whiteson, S., de Freitas, N.: Lipnet: Sentence-level lipreading. arXiv preprint arXiv:1611.01599 (2016)"},{"issue":"2","key":"9_CR2","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1016\/j.inffus.2003.04.001","volume":"5","author":"S Bengio","year":"2004","unstructured":"Bengio, S.: Multimodal speech processing using asynchronous hidden markov models. Inform. Fusion 5(2), 81\u201389 (2004)","journal-title":"Inform. Fusion"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Brand, M., Oliver, N., Pentland, A.: Coupled hidden markov models for complex action recognition. In: Proceedings IEEE Computer Society Conference on Computer vision and pattern recognition, pp. 994\u2013999. IEEE (1997)","DOI":"10.1109\/CVPR.1997.609450"},{"issue":"5","key":"9_CR4","doi-asserted-by":"crossref","first-page":"2421","DOI":"10.1121\/1.2229005","volume":"120","author":"M Cooke","year":"2006","unstructured":"Cooke, M., Barker, J., Cunningham, S., Shao, X.: An audio-visual corpus for speech perception and automatic speech recognition. J. Acoust. Soc. Am. 120(5), 2421\u20132424 (2006)","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"9_CR5","doi-asserted-by":"crossref","first-page":"141","DOI":"10.1109\/6046.865479","volume":"2","author":"S Dupont","year":"2000","unstructured":"Dupont, S., Luettin, J.: Audio-visual speech modeling for continuous speech recognition. IEEE Trans. Multimedia 2(3), 141\u2013151 (2000)","journal-title":"IEEE Trans. Multimedia"},{"issue":"4","key":"9_CR6","doi-asserted-by":"crossref","first-page":"469","DOI":"10.1109\/TVCG.2005.54","volume":"11","author":"J Ebling","year":"2005","unstructured":"Ebling, J., Scheuermann, G.: Clifford fourier transform on vector fields. IEEE Trans. Vis. Comput. Graph. 11(4), 469\u2013479 (2005)","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"key":"9_CR7","doi-asserted-by":"crossref","unstructured":"Gehring, J., Miao, Y., Metze, F., Waibel, A.: Extracting deep bottleneck features using stacked auto-encoders. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 3377\u20133381. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6638284"},{"key":"9_CR8","doi-asserted-by":"crossref","unstructured":"Graves, A.: Supervised Sequence Labelling with Recurrent Neural Networks. SCI, vol. 385, pp. 15\u201335. Springer, Heidelberg (2012)","DOI":"10.1007\/978-3-642-24797-2_3"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376. ACM (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"9_CR10","unstructured":"Graves, A., Jaitly, N.: Towards end-to-end speech recognition with recurrent neural networks. In: ICML, vol. 14, pp. 1764\u20131772 (2014)"},{"key":"9_CR11","unstructured":"Hannun, A., Case, C., Casper, J., Catanzaro, B., Diamos, G., Elsen, E., Prenger, R., Satheesh, S., Sengupta, S., Coates, A., et al.: Deep speech: scaling up end-to-end speech recognition. arXiv preprint arXiv:1412.5567 (2014)"},{"key":"9_CR12","doi-asserted-by":"crossref","unstructured":"Hermansky, H., Ellis, D.P., Sharma, S.: Tandem connectionist feature extraction for conventional hmm systems. In: Proceedings of the 2000 IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP 2000, vol. 3, pp. 1635\u20131638. IEEE (2000)","DOI":"10.1109\/ICASSP.2000.862024"},{"issue":"8","key":"9_CR13","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Huang, J., Kingsbury, B.: Audio-visual deep learning for noise robust speech recognition. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 7596\u20137599. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639140"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Huang, J., Potamianos, G., Neti, C.: Improving audio-visual speech recognition with an infrared headset. In: AVSP 2003-International Conference on Audio-Visual Speech Processing (2003)","DOI":"10.21437\/Eurospeech.2003-410"},{"issue":"9","key":"9_CR16","doi-asserted-by":"crossref","first-page":"1635","DOI":"10.1109\/JPROC.2015.2459017","volume":"103","author":"AK Katsaggelos","year":"2015","unstructured":"Katsaggelos, A.K., Bahaadini, S., Molina, R.: Audiovisual fusion: challenges and new approaches. Proc. IEEE 103(9), 1635\u20131653 (2015)","journal-title":"Proc. IEEE"},{"key":"9_CR17","doi-asserted-by":"crossref","unstructured":"Kazemi, V., Sullivan, J.: One millisecond face alignment with an ensemble of regression trees. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1867\u20131874 (2014)","DOI":"10.1109\/CVPR.2014.241"},{"key":"9_CR18","unstructured":"Miao, Y.: Kaldi+pdnn: building dnn-based asr systems with kaldi and pdnn. arXiv preprint arXiv:1401.6984 (2014)"},{"key":"9_CR19","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Metze, F.: Eesen: End-to-end speech recognition using deep rnn models and wfst-based decoding. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 167\u2013174. IEEE (2015)","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Mohammadzade, H., Bruton, L.T.: A simultaneous div-curl 2D clifford fourier transform filter for enhancing vortices, sinks and sources in sampled 2D vector field images. In: IEEE International Symposium on Circuits and Systems, ISCAS 2007, pp. 821\u2013824. IEEE (2007)","DOI":"10.1109\/ISCAS.2007.378032"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Mroueh, Y., Marcheret, E., Goel, V.: Deep multimodal learning for audio-visual speech recognition. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2130\u20132134. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178347"},{"key":"9_CR22","unstructured":"Ngiam, J., Khosla, A., Kim, M., Nam, J., Lee, H., Ng, A.Y.: Multimodal deep learning. In: Proceedings of the 28th International Conference on Machine Learning (ICML 2011), pp. 689\u2013696 (2011)"},{"issue":"4","key":"9_CR23","doi-asserted-by":"crossref","first-page":"722","DOI":"10.1007\/s10489-014-0629-7","volume":"42","author":"K Noda","year":"2015","unstructured":"Noda, K., Yamaguchi, Y., Nakadai, K., Okuno, H.G., Ogata, T.: Audio-visual speech recognition using deep learning. Appl. Intell. 42(4), 722\u2013737 (2015)","journal-title":"Appl. Intell."},{"issue":"9","key":"9_CR24","doi-asserted-by":"crossref","first-page":"1306","DOI":"10.1109\/JPROC.2003.817150","volume":"91","author":"G Potamianos","year":"2003","unstructured":"Potamianos, G., Neti, C., Gravier, G., Garg, A., Senior, A.W.: Recent advances in the automatic recognition of audiovisual speech. Proc. IEEE 91(9), 1306\u20131326 (2003)","journal-title":"Proc. IEEE"},{"key":"9_CR25","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., et al.: The kaldi speech recognition toolkit. In: IEEE 2011 Workshop on Automatic Speech Recognition and Understanding. No. EPFL-CONF-192584. IEEE Signal Processing Society (2011)"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Vesel\u1ef3, K., Ghoshal, A., Burget, L., Povey, D.: Sequence-discriminative training of deep neural networks. In: INTERSPEECH, pp. 2345\u20132349 (2013)","DOI":"10.21437\/Interspeech.2013-548"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Wand, M., Koutn\u00edk, J., Schmidhuber, J.: Lipreading with long short-term memory. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6115\u20136119. IEEE (2016)","DOI":"10.1109\/ICASSP.2016.7472852"},{"key":"9_CR28","doi-asserted-by":"crossref","unstructured":"Yu, D., Seltzer, M.L.: Improved bottleneck features using pretrained deep neural networks. In: Interspeech, vol. 237, p. 240 (2011)","DOI":"10.21437\/Interspeech.2011-91"}],"container-title":["Lecture Notes in Computer Science","Multimodal Pattern Recognition of Social Signals in Human-Computer-Interaction"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-59259-6_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,23]],"date-time":"2023-08-23T20:39:11Z","timestamp":1692823151000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-59259-6_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319592589","9783319592596"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-59259-6_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2017]]}}}