{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T02:26:15Z","timestamp":1783650375602,"version":"3.55.0"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,6,6]]},"DOI":"10.1109\/icassp39728.2021.9415063","type":"proceedings-article","created":{"date-parts":[[2021,5,13]],"date-time":"2021-05-13T15:53:45Z","timestamp":1620921225000},"page":"7608-7612","source":"Crossref","is-referenced-by-count":82,"title":["Towards Practical Lipreading with Distilled and Efficient Models"],"prefix":"10.1109","author":[{"given":"Pingchuan","family":"Ma","sequence":"first","affiliation":[{"name":"Imperial College London,Department of Computing,UK"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Brais","family":"Martinez","sequence":"additional","affiliation":[{"name":"Samsung AI Research Center,Cambridge,UK"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stavros","family":"Petridis","sequence":"additional","affiliation":[{"name":"Imperial College London,Department of Computing,UK"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maja","family":"Pantic","sequence":"additional","affiliation":[{"name":"Imperial College London,Department of Computing,UK"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00756"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"ref31","first-page":"276","article-title":"Multi-grained spatio-temporal modeling for lip-reading","author":"wang","year":"2019","journal-title":"BMVC"},{"key":"ref30","article-title":"Mixup: Beyond empirical risk minimization","author":"zhang","year":"2018","journal-title":"ICLRE"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639643"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2618"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3273"},{"key":"ref13","article-title":"Distilling the knowledge in a neural network","author":"hinton","year":"2014","journal-title":"NIPS Workshop"},{"key":"ref14","first-page":"1602","article-title":"Born-again neural networks","author":"furlanello","year":"2018","journal-title":"ICML"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015628"},{"key":"ref16","article-title":"MobileNets: Efficient convolutional neural networks for mobile vision applications","author":"howard","year":"2017","journal-title":"CoRR"},{"key":"ref17","first-page":"122","article-title":"ShufflenetV2: practical guidelines for efficient CNN architecture design","author":"ma","year":"2018","journal-title":"ECCV"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref19","article-title":"Training binary neural networks with real-to-binary convolutions","author":"martinez","year":"2020","journal-title":"ICLRE"},{"key":"ref28","first-page":"1755","article-title":"Dlib-ml: A machine learning toolkit","volume":"10","author":"king","year":"2009","journal-title":"J Mach Learn Res"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472852"},{"key":"ref27","article-title":"Can we read speech beyond the lips? rethinking RoI selection for deep visual speech recognition","author":"zhang","year":"2020","journal-title":"FG"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.5244\/C.31.161"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"3652","DOI":"10.21437\/Interspeech.2017-85","article-title":"Combining residual networks with LSTMs for lipreading","author":"stafylakis","year":"2017","journal-title":"InterSpeech"},{"key":"ref29","article-title":"Decoupled weight decay regularization","author":"loshchilov","year":"2019","journal-title":"ICLRE"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"36","DOI":"10.21437\/AVSP.2017-8","article-title":"End-to-end audiovisual fusion with LSTMs","author":"petridis","year":"2017","journal-title":"AVSPN"},{"key":"ref8","article-title":"Deep audio-visual speech recognition","author":"afouras","year":"2018","journal-title":"CoRR"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1669"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461596"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.367"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952625"},{"key":"ref20","first-page":"1","article-title":"LRW-1000: A naturally-distributed large-scale benchmark for lip reading in the wild","author":"yang","year":"2019","journal-title":"FG"},{"key":"ref22","article-title":"Label refinery: Improving ImageNet classification through label progression","author":"bagherinezhad","year":"2018"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053841"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461326"},{"key":"ref23","first-page":"87","article-title":"Lip reading in the wild","volume":"10112","author":"chung","year":"2016","journal-title":"ACCV"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2018.10.003"},{"key":"ref25","first-page":"269","article-title":"Learning spatio-temporal features with two-stream deep 3d cnns for lipreading","author":"weng","year":"2019","journal-title":"BMVC"}],"event":{"name":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Toronto, ON, Canada","start":{"date-parts":[[2021,6,6]]},"end":{"date-parts":[[2021,6,11]]}},"container-title":["ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9413349\/9413350\/09415063.pdf?arnumber=9415063","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,2]],"date-time":"2022-08-02T20:19:38Z","timestamp":1659471578000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9415063\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,6]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/icassp39728.2021.9415063","relation":{},"subject":[],"published":{"date-parts":[[2021,6,6]]}}}