{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T21:04:13Z","timestamp":1774991053358,"version":"3.50.1"},"reference-count":24,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T00:00:00Z","timestamp":1721001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T00:00:00Z","timestamp":1721001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,7,15]]},"DOI":"10.1109\/icmew63481.2024.10645400","type":"proceedings-article","created":{"date-parts":[[2024,8,29]],"date-time":"2024-08-29T17:43:36Z","timestamp":1724953416000},"page":"1-6","source":"Crossref","is-referenced-by-count":4,"title":["Enhancing Lip Reading with Multi-Scale Video and Multi-Encoder"],"prefix":"10.1109","author":[{"given":"He","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xian,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengcheng","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xian,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xucheng","family":"Wan","sequence":"additional","affiliation":[{"name":"IT Innovation and Research Center, Huawei Technologies"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huan","family":"Zhou","sequence":"additional","affiliation":[{"name":"IT Innovation and Research Center, Huawei Technologies"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Xie","sequence":"additional","affiliation":[{"name":"School of Computer Science, Northwestern Polytechnical University,Audio, Speech and Language Processing Group (ASLP@NPU),Xian,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Achieving Human Parity in Conversational Speech Recognition","author":"Xiong","year":"2016","journal-title":"arXiv preprint"},{"key":"ref2","article-title":"At-tention is All You Need","volume-title":"Proc. NIPS","volume":"30","author":"Vaswani","year":"2017"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.3390\/app10207263"},{"key":"ref4","first-page":"796","article-title":"Audio-visual Speech Recognition is Worth 32 X 32 X 8 Voxels","volume-title":"Proc. ASRU","author":"Serdyuk","year":"2021"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-10920"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICFTIC57696.2022.10075247"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446532"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414567"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096889"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/icassp43922.2022.9746683"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/icassp49357.2023.10094836"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10447462"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10483"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095796"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446769"},{"key":"ref17","article-title":"The NPU-ASL P-LiAuto System Description for Visual Speech Recognition in CNVSRC 2023","author":"Wang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref18","first-page":"17627","article-title":"Branchformer: Parallel MLP- Attention Architectures to Capture Local and Global Context for Speech Recognition and Understanding","volume-title":"Proc. ICML","author":"Peng","year":"2022"},{"key":"ref19","first-page":"84","article-title":"E-branchformer: Branch-former with Enhanced Merging for Speech Recognition","volume-title":"Proc. SLT","author":"Kim","year":"2023"},{"key":"ref20","first-page":"347","article-title":"A Post-processing System to Yield Reduced Word Error Rates: Recognizer output voting error reduction (ROVER)","volume-title":"Proc. ASRU","author":"Jonathan","year":"1997"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/FG47880.2020.00134"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref24","first-page":"2207","article-title":"Espnet: End-to-End Speech Processing Toolkit","volume-title":"Proc. Inter-speech","author":"WWatanabe","year":"2018"}],"event":{"name":"2024 IEEE International Conference on Multimedia and Expo Workshops (ICMEW)","location":"Niagara Falls, ON, Canada","start":{"date-parts":[[2024,7,15]]},"end":{"date-parts":[[2024,7,19]]}},"container-title":["2024 IEEE International Conference on Multimedia and Expo Workshops (ICMEW)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10645349\/10645352\/10645400.pdf?arnumber=10645400","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,31]],"date-time":"2024-08-31T05:21:38Z","timestamp":1725081698000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10645400\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,15]]},"references-count":24,"URL":"https:\/\/doi.org\/10.1109\/icmew63481.2024.10645400","relation":{},"subject":[],"published":{"date-parts":[[2024,7,15]]}}}