{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:38:06Z","timestamp":1783557486812,"version":"3.55.0"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"4","content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2015,6]]},"DOI":"10.1007\/s10489-014-0629-7","type":"journal-article","created":{"date-parts":[[2014,12,19]],"date-time":"2014-12-19T09:49:46Z","timestamp":1418982586000},"page":"722-737","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":459,"title":["Audio-visual speech recognition using deep learning"],"prefix":"10.1007","volume":"42","author":[{"given":"Kuniaki","family":"Noda","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuki","family":"Yamaguchi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kazuhiro","family":"Nakadai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hiroshi G.","family":"Okuno","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tetsuya","family":"Ogata","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2014,12,20]]},"reference":[{"key":"629_CR1","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid O, Jiang H. (2013) Rapid and effective speaker adaptation of convolutional neural network based models for speech recognition. In: Proceedings of the 14th Annual Conference of the International Speech Communication Association. Lyon, France","DOI":"10.21437\/Interspeech.2013-336"},{"key":"629_CR2","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid O, rahman Mohamed A, Jiang H, Penn G (2012) Applying convolutional neural networks concepts to hybrid NN-HMM model for speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech,and Signal Processing, Kyoto, pp 4277\u20134280","DOI":"10.1109\/ICASSP.2012.6288864"},{"key":"629_CR3","unstructured":"Aleksic PS, Katsaggelos AK (2004) Comparison of low- and high-level visual features for audio-visual continuous automatic speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, vol 5, Montreal, pp 917\u2013920"},{"key":"629_CR4","unstructured":"Barker J, Berthommier F (1999) Evidence of correlation between acoustic and visual features of speech. In: Proceedings of the 14th International Congress of Phonetic Sciences, San Francisco , pp 5\u20139"},{"issue":"1","key":"629_CR5","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/2200000006","volume":"2","author":"Y Bengio","year":"2009","unstructured":"Bengio Y (2009) Learning deep architectures for AI. Found Trends Mach Learn 2(1):1\u2013127","journal-title":"Found Trends Mach Learn"},{"key":"629_CR6","doi-asserted-by":"crossref","unstructured":"Bourlard H, Dupont S (1996) A new ASR approach based on independent processing and recombination of partial frequency bands. In: Proceedings of the 4th International Conference on Spoken Language Processing, vol 1, Philadelphia, pp 426\u2013429","DOI":"10.1109\/ICSLP.1996.607145"},{"key":"629_CR7","unstructured":"Bourlard H, Dupont S, Ris C (1996) Multi-stream speech recognition.IDIAP research report"},{"key":"629_CR8","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4615-3210-1","volume-title":"Connectionist speech recognition: a hybrid approach","author":"Ha Bourlard","year":"1994","unstructured":"Bourlard H a, Morgan N (1994) Connectionist speech recognition: a hybrid approach. Springer US, Boston"},{"key":"629_CR9","unstructured":"Brooke N, Petajan ED (1986) Seeing speech: Investigations into the synthesis and recognition of visible speech movements using automatic image processing and computer graphics. In: Proceedings of the International Conference on Speech Input and Output, Techniques and Applications, London, pp 104\u2013109"},{"key":"629_CR10","unstructured":"Coates A, Huval B, Wang T, Wu DJ, Ng AY, Catanzaro B (2013) Deep learning with COTS HPC. In: Proceedings of the 30th international conference on machine learning, Atlanta, pp 1337\u20131345"},{"issue":"6","key":"629_CR11","doi-asserted-by":"crossref","first-page":"681","DOI":"10.1109\/34.927467","volume":"23","author":"T Cootes","year":"2001","unstructured":"Cootes T, Edwards G, Taylor C (2001) Active appearance models. IEEE Trans Pattern Anal Mach Intell 23(6):681\u2013685","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"1","key":"629_CR12","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"GE Dahl","year":"2012","unstructured":"Dahl GE, Acero A (2012) Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans Audio Speech Lang Process 20(1):30\u201342","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"629_CR13","doi-asserted-by":"crossref","unstructured":"Feng X, Zhang Y, Glass J (2014) Speech feature denoising and dereverberation via deep autoencoders for noisy reverberant speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, Florence, pp 1759\u20131763","DOI":"10.1109\/ICASSP.2014.6853900"},{"key":"629_CR14","doi-asserted-by":"crossref","unstructured":"Gurban M, Thiran JP, Drugman T, Dutoit T (2008) Dynamic modality weighting for multi-stream HMMs in audio-visual speech recognition. In: Proceedings of the 10th International Conference on Multimodal Interfaces, Chania, pp 237\u2013 240","DOI":"10.1145\/1452392.1452442"},{"key":"629_CR15","doi-asserted-by":"crossref","unstructured":"Heckmann M, Kroschel K, Savariaux C (2002) DCT-based video features for audio-visual speech recognition. In: Proceedings of the 7th International Conference on Spoken Language Processing, vol 3, Denver, pp 1925\u20131928","DOI":"10.21437\/ICSLP.2002-434"},{"key":"629_CR16","doi-asserted-by":"crossref","unstructured":"Hermansky H, Ellis D, Sharma S (2000) Tandem connectionist feature extraction for conventional HMM systems. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, vol 3, Istanbul, pp 1635\u20131638","DOI":"10.1109\/ICASSP.2000.862024"},{"key":"629_CR17","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D, Dahl G, Mohamed A, Jaitly N, Senior A, Vanhoucke V, Nguyen P, Sainath T, Kingsbury B (2012) Deep neural networks for acoustic modeling in speech recognition. IEEE Signal Proc Mag 29:82\u201397","journal-title":"IEEE Signal Proc Mag"},{"issue":"5786","key":"629_CR18","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Salakhutdinov RR (2006) Reducing the dimensionality of data with neural networks. Science 313(5786):504\u20137","journal-title":"Science"},{"key":"629_CR19","doi-asserted-by":"crossref","unstructured":"Huang J, Kingsbury B (2013) Audio-visual deep learning for noise robust speech recognition. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, Vancouver, pp 7596\u20137599","DOI":"10.1109\/ICASSP.2013.6639140"},{"key":"629_CR20","doi-asserted-by":"crossref","unstructured":"Janin A, Ellis D, Morgan N (1999) Multi-stream speech recognition: Ready for prime time? In: Proceedings of the 6th European Conference on Speech Communication and Technology. Budapest, Hungary","DOI":"10.21437\/Eurospeech.1999-152"},{"key":"629_CR21","unstructured":"Krizhevsky A, Hinton GE (2011) Using very deep autoencoders for content-based image retrieval. In: Proceedings of the 19th European Symposium on Artificial Neural Networks. Bruges, Belgium"},{"key":"629_CR22","unstructured":"Krizhevsky A, Sutskever I, Hinton G (2012) Imagenet classification with deep convolutional neural networks. Advances in Neural Information Processing Systems"},{"key":"629_CR23","doi-asserted-by":"crossref","unstructured":"Kuwabara H, Takeda K, Sagisaka Y, Katagiri S, Morikawa S, Watanabe T (1989) Construction of a large-scale Japanese speech database and its management system. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, Glasgow, pp 560\u2013563","DOI":"10.1109\/ICASSP.1989.266488"},{"key":"629_CR24","unstructured":"Lan Y, Theobald BJ, Harvey R, Ong EJ, Bowden R (2010) Improving visual features for lip-reading. In: Proceedings of the International Conference on Auditory-Visual Speech Processing. Hakone,Japan"},{"key":"629_CR25","unstructured":"Le QV, Ranzato M, Monga R, Devin M, Chen K, Corrado GS, Dean J, Ng AY (2012) Building high-level features using large scale unsupervised learning. In: Proceedings of the 29th International Conference on Machine Learning, Edinburgh, pp 81\u201388"},{"key":"629_CR26","doi-asserted-by":"crossref","unstructured":"LeCun Y, Bottou L (2004) Learning methods for generic object recognition with invariance to pose and lighting. In: Proceedings of the IEEE Computer Society Conference on Computer Vision and Pattern Recognition, vol 2, Washington, pp 97\u2013104","DOI":"10.1109\/CVPR.2004.1315150"},{"issue":"11","key":"629_CR27","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324","journal-title":"Proc IEEE"},{"key":"629_CR28","doi-asserted-by":"crossref","unstructured":"Lee H, Grosse R, Ranganath R, Ng AY (2009) Convolutional deep belief networks for scalable unsupervised learning of hierarchical representations. In: Proceedings of the 26th International Conference on Machine Learning, Montreal, pp 609\u2013 616","DOI":"10.1145\/1553374.1553453"},{"key":"629_CR29","unstructured":"Lee H, Pham P, Largman Y, Ng AY (2009) Unsupervised feature learning for audio classification using convolutional deep belief networks. In: Proceedings of the Advances in Neural Information Processing Systems 22, Vancouver, pp 1096\u20131104"},{"issue":"1","key":"629_CR30","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1016\/S0167-8655(98)00120-2","volume":"20","author":"B Lerner","year":"1999","unstructured":"Lerner B, Guterman H, Aladjem M, Dinstein I (1999) A comparative study of neural network based feature extraction paradigms. Pattern Recogn Lett 20(1):7\u201314","journal-title":"Pattern Recogn Lett"},{"key":"629_CR31","doi-asserted-by":"crossref","unstructured":"Luettin J, Thacker N, Beet S (1996) Visual speech recognition using active shape models and hidden Markov models. In: Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, vol 2, Atlanta , pp 817\u2013820","DOI":"10.1109\/ICASSP.1996.543246"},{"key":"629_CR32","unstructured":"Maas AL, O\u2019Neil TM, Hannun AY, Ng AY (2013) Recurrent neural network feature enhancement: The 2nd chime challenge. In: Proceedings of the 2nd International Workshop on Machine Listening in Multisource Environments.Vancouver, Canada"},{"key":"629_CR33","unstructured":"Martens J (2010) Deep learning via Hessian-free optimization. In: Proceedings of the 27th International Conference on Learning, Machine, Haifa, pp 735\u2013742"},{"issue":"2","key":"629_CR34","doi-asserted-by":"crossref","first-page":"198","DOI":"10.1109\/34.982900","volume":"24","author":"I Matthews","year":"2002","unstructured":"Matthews I, Cootes T, Bangham J, Cox S, Harvey R (2002) Extraction of visual features for lipreading. IEEE Trans Pattern Anal Mach Intell 24(2):198\u2013213","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"629_CR35","doi-asserted-by":"crossref","unstructured":"Matthews I, Potamianos G, Neti C, Luettin J (2001) A comparison of model and transform-based visual features for audio-visual LVCSR. In: Proceedings of the IEEE International Conference on Multimedia and Expo. Tokyo, Japan","DOI":"10.1109\/ICME.2001.1237849"},{"issue":"1","key":"629_CR36","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A Mohamed","year":"2012","unstructured":"Mohamed A, Dahl GE, Hinton GE (2012) Acoustic modeling using deep belief networks. IEEE Trans Audio Speech Lang Process 20(1):14\u201322","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"629_CR37","unstructured":"Ngiam J, Khosla A, Kim M, Nam J, Lee H, Ng AY (2011) Multimodal deep learning. In: Proceedings of the 28th International Conference on Machine Learning"},{"key":"629_CR38","unstructured":"NVIDIA Corporation (2014) CUBLAS library version 6.0 user guide. CUDA Toolkit Documentation"},{"key":"629_CR39","doi-asserted-by":"crossref","unstructured":"Palaz D, Collobert R, Magimai.-Doss M (2013) Estimating phoneme class conditional probabilities from raw speech signal using convolutional neural networks. In: Proceedings of the 14th Annual Conference of the International Speech Communication Association. Lyon, France","DOI":"10.21437\/Interspeech.2013-438"},{"issue":"1","key":"629_CR40","doi-asserted-by":"crossref","first-page":"147","DOI":"10.1162\/neco.1994.6.1.147","volume":"6","author":"B Pearlmutter","year":"1994","unstructured":"Pearlmutter B (1994) Fast exact multiplication by the Hessian. Neural Comput 6(1):147\u2013160","journal-title":"Neural Comput"},{"key":"629_CR41","doi-asserted-by":"crossref","unstructured":"Renals S, Morgan N, Member S, Bourlard H, Cohen M, Franco H (1994) Connectionist probability estimators in HMM speech recognition 2(1):161\u2013174","DOI":"10.1109\/89.260359"},{"key":"629_CR42","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1007\/978-3-662-13015-5_14","volume-title":"Speechreading by Humans and Machines","author":"J Robert-Ribes","year":"1996","unstructured":"Robert-Ribes J, Piquemal M, Schwartz JL, Escudier P (1996) Exploiting sensor fusion architectures and stimuli complementarity in av speech recognition. In: Stork D, Hennecke M (eds) Speechreading by Humans and Machines. Springer, Berlin Heidelberg, pp 193\u2013210"},{"key":"629_CR43","doi-asserted-by":"crossref","unstructured":"Sainath TN, Kingsbury B, Ramabhadran B (2012) Auto-encoder bottleneck features using deep belief networks. In:Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing, Kyoto, pp 4153\u20134156","DOI":"10.1109\/ICASSP.2012.6288833"},{"key":"629_CR44","doi-asserted-by":"crossref","unstructured":"Scanlon P, Reilly R (2001) Feature analysis for automatic speechreading. In: Proceedings of the IEEE 4th Workshop on Processing, Multimedia Signal, Cannes, pp 625\u2013630","DOI":"10.1109\/MMSP.2001.962802"},{"issue":"7","key":"629_CR45","doi-asserted-by":"crossref","first-page":"1723","DOI":"10.1162\/08997660260028683","volume":"14","author":"NN Schraudolph","year":"2002","unstructured":"Schraudolph NN (2002) Fast curvature matrix-vector products for second-order gradient descent. Neural Comput 14(7):1723\u201338","journal-title":"Neural Comput"},{"key":"629_CR46","unstructured":"Slaney M (1998) Auditory toolbox: A MATLAB toolbox for auditory modeling work version 2. Interval research corproation"},{"key":"629_CR47","unstructured":"Sutskever I, Martens J, Hinton G (2011) Generating text with recurrent neural networks. In: Proceedings of the 28th International Conference on Machine Learning, Bellevue, pp 1017\u20131024"},{"key":"629_CR48","doi-asserted-by":"crossref","unstructured":"Vincent P, Larochelle H, Bengio Y, Manzagol PA (2008) Extracting and composing robust features with denoising autoencoders. In: Proceedings of the 25th international conference on Machine learning, New York, pp 1096\u20131103","DOI":"10.1145\/1390156.1390294"},{"key":"629_CR49","first-page":"3371","volume":"11","author":"P Vincent","year":"2010","unstructured":"Vincent P, Larochelle H, Lajoie I, Bengio Y, Manzagol PA (2010) Stacked denoising autoencoders: Learning useful representations in a deep network with a local denoising criterion. J Mach Learn Res 11:3371\u20133408","journal-title":"J Mach Learn Res"},{"key":"629_CR50","doi-asserted-by":"crossref","first-page":"23","DOI":"10.1016\/S0167-6393(98)00048-X","volume":"26","author":"H Yehia","year":"1998","unstructured":"Yehia H, Rubin P, Vatikiotis-Bateson E (1998) Quantitative association of vocal-tract and facial behavior. Speech Comm 26:23\u201343","journal-title":"Speech Comm"},{"key":"629_CR51","doi-asserted-by":"crossref","unstructured":"Yoshida T, Nakadai K, Okuno HG (2009) Automatic speech recognition improved by two-layered audio-visual integration for robot audition. In: Proceedings of the 9th IEEE-RAS International Conference on Humanoid Robots, Paris, pp 604\u2013609","DOI":"10.1109\/ICHR.2009.5379586"},{"key":"629_CR52","unstructured":"Young S, Evermann G, Gales M, Hain T, Liu XA, Moore G, Odell J, Ollason D, Povey D, Valtchev V, Woodland P (2009) The HTK Book (for HTK Version 3.4),.Cambridge University Engineering Department"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-014-0629-7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,30]],"date-time":"2023-07-30T20:02:49Z","timestamp":1690747369000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10489-014-0629-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,12,20]]},"references-count":52,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2015,6]]}},"alternative-id":["629"],"URL":"https:\/\/doi.org\/10.1007\/s10489-014-0629-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,12,20]]}}}