{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T22:34:41Z","timestamp":1762036481099,"version":"build-2065373602"},"reference-count":30,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,19]]},"DOI":"10.1109\/slt48900.2021.9383466","type":"proceedings-article","created":{"date-parts":[[2021,3,25]],"date-time":"2021-03-25T20:46:54Z","timestamp":1616705214000},"page":"621-628","source":"Crossref","is-referenced-by-count":5,"title":["Listen, Look and Deliberate: Visual Context-Aware Speech Recognition Using Pre-Trained Text-Video Representations"],"prefix":"10.1109","author":[{"given":"Shahram","family":"Ghorbani","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yashesh","family":"Gaur","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinyu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref30","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"3rd International Conference on Learning Representations ICLR 2015"},{"key":"ref10","article-title":"Multimodal abstractive summarization of open-domain videos","author":"libovick?","year":"2018","journal-title":"Proceedings of the Workshop on Visually Grounded Interaction and Language (ViGIL) NIPS"},{"key":"ref11","article-title":"Deep audio-visual speech recognition","author":"afouras","year":"2018","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"article-title":"Lipnet: End-to-end sentence-level lipreading","year":"2016","author":"assael","key":"ref12"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-412"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1329"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2020.emnlp-main.161","article-title":"Hero: Hierarchical encoder for video+ language omni-representation pre-training","author":"li","year":"2020"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_40"},{"key":"ref18","first-page":"2514","article-title":"Semantic speech retrieval with a visually grounded model of untranscribed speech","author":"kamper","year":"2018","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref28","article-title":"Empirical evaluation of gated recurrent neural networks on sequence modeling","author":"chung","year":"2014","journal-title":"NIPS 2014 Deep Learning Workshop"},{"article-title":"Analyzing utility of visual context in multimodal speech recognition under noisy conditions","year":"2019","author":"srinivasan","key":"ref4"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-2012"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953112"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639551"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682750"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682583"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1422"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462439"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101102"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053397"},{"key":"ref20","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2015","journal-title":"3rd International Conference on Learning Representations ICLR 2015"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053606"},{"key":"ref21","first-page":"1784","article-title":"Deliberation networks: Sequence generation beyond one-pass decoding","author":"xia","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682801"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1341"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00685"},{"key":"ref25","article-title":"How2: A large-scale dataset for multimodal language understanding","author":"sanabria","year":"2018","journal-title":"NeurIPS"}],"event":{"name":"2021 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2021,1,19]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,1,22]]}},"container-title":["2021 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9383468\/9383452\/09383466.pdf?arnumber=9383466","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,22]],"date-time":"2022-12-22T13:16:13Z","timestamp":1671714973000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9383466\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,19]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/slt48900.2021.9383466","relation":{},"subject":[],"published":{"date-parts":[[2021,1,19]]}}}