{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:39:03Z","timestamp":1784738343627,"version":"3.55.0"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2022,10,24]],"date-time":"2022-10-24T00:00:00Z","timestamp":1666569600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,10,24]],"date-time":"2022-10-24T00:00:00Z","timestamp":1666569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Nat Mach Intell"],"DOI":"10.1038\/s42256-022-00550-z","type":"journal-article","created":{"date-parts":[[2022,10,24]],"date-time":"2022-10-24T12:06:35Z","timestamp":1666613195000},"page":"930-939","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":145,"title":["Visual speech recognition for multiple languages in the wild"],"prefix":"10.1038","volume":"4","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3752-0803","authenticated-orcid":false,"given":"Pingchuan","family":"Ma","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stavros","family":"Petridis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maja","family":"Pantic","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,10,24]]},"reference":[{"key":"550_CR1","doi-asserted-by":"publisher","first-page":"1306","DOI":"10.1109\/JPROC.2003.817150","volume":"91","author":"G Potamianos","year":"2003","unstructured":"Potamianos, G., Neti, C., Gravier, G., Garg, A. & Senior, A. W. Recent advances in the automatic recognition of audiovisual speech. Proc. IEEE 91, 1306\u20131326 (2003).","journal-title":"Proc. IEEE"},{"key":"550_CR2","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1109\/6046.865479","volume":"2","author":"S Dupont","year":"2000","unstructured":"Dupont, S. & Luettin, J. Audio-visual speech modeling for continuous speech recognition. IEEE Trans. Multimedia 2, 141\u2013151 (2000).","journal-title":"IEEE Trans. Multimedia"},{"key":"550_CR3","doi-asserted-by":"crossref","unstructured":"Chung, J. S., Senior, A., Vinyals, O. & Zisserman, A. Lip reading sentences in the wild. In Proc. 30th IEEE\/CVF Conference on Computer Vision and Pattern Recognition 3444\u20133453 (IEEE, 2017).","DOI":"10.1109\/CVPR.2017.367"},{"key":"550_CR4","doi-asserted-by":"publisher","unstructured":"Afouras, T., Chung, J. S., Senior, A., Vinyals, O. & Zisserman, A. Deep audio-visual speech recognition. In IEEE Transactions on Pattern Analysis and Machine Intelligence, 1 (IEEE, 2018); https:\/\/doi.org\/10.1109\/TPAMI.2018.2889052","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"550_CR5","unstructured":"Shillingford, B. et al. Large-scale visual speech recognition. In Proc. 20th Annual Conference of International Speech Communication Association 4135\u20134139 (ISCA, 2019)."},{"key":"550_CR6","doi-asserted-by":"crossref","unstructured":"Serdyuk, D., Braga, O. & Siohan, O. Audio-visual speech recognition is worth 32\u2009\u00d7\u200932\u2009\u00d7\u20098 voxels. In Proc. IEEE Automatic Speech Recognition and Understanding Workshop 796\u2013802 (IEEE, 2021).","DOI":"10.1109\/ASRU51503.2021.9688191"},{"key":"550_CR7","doi-asserted-by":"crossref","unstructured":"Zhang, X. et al. Understanding pictograph with facial features: end-to-end sentence-level lip reading of Chinese. In Proc. 33rd AAAI Conference on Artificial Intelligence 9211\u20139218 (AAAI, 2019).","DOI":"10.1609\/aaai.v33i01.33019211"},{"key":"550_CR8","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Xu, R. & Song, M. A cascade sequence-to-sequence model for Chinese Mandarin lip reading. In Proc. 1st ACM International Conference on Multimedia in Asia 1\u20136 (ACM, 2019).","DOI":"10.1145\/3338533.3366579"},{"key":"550_CR9","doi-asserted-by":"crossref","unstructured":"Ma, S., Wang, S. & Lin, X. A transformer-based model for sentence-level Chinese Mandarin lipreading. In Proc. 5th IEEE International Conference on Data Science in Cyberspace 78\u201381 (IEEE, 2020).","DOI":"10.1109\/DSC50466.2020.00020"},{"key":"550_CR10","doi-asserted-by":"crossref","unstructured":"Ma, P., Petridis, S. & Pantic, M. End-to-end audio-visual speech recognition with conformers. In Proc. 46th IEEE International Conference on Acoustics, Speech and Signal Processing 7613\u20137617 (IEEE, 2021).","DOI":"10.1109\/ICASSP39728.2021.9414567"},{"key":"550_CR11","doi-asserted-by":"crossref","unstructured":"Gulati, A. et al. Conformer: convolution-augmented transformer for speech recognition. In Proc. 21st Annual Conference of International Speech Communication Association 5036\u20135040 (ISCA, 2020).","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"550_CR12","doi-asserted-by":"crossref","unstructured":"Makino, T. et al. Recurrent neural network transducer for audio-visual speech recognition. In Proc. IEEE Automatic Speech Recognition and Understanding Workshop 905\u2013912 (IEEE, 2019).","DOI":"10.1109\/ASRU46091.2019.9004036"},{"key":"550_CR13","doi-asserted-by":"publisher","first-page":"746","DOI":"10.1038\/264746a0","volume":"264","author":"H McGurk","year":"1976","unstructured":"McGurk, H. & MacDonald, J. Hearing lips and seeing voices. Nature 264, 746\u2013748 (1976).","journal-title":"Nature"},{"key":"550_CR14","doi-asserted-by":"publisher","first-page":"212","DOI":"10.1121\/1.1907309","volume":"26","author":"WH Sumby","year":"1954","unstructured":"Sumby, W. H. & Pollack, I. Visual contribution to speech intelligibility in noise. J. Acoust. Soc. Am. 26, 212\u2013215 (1954).","journal-title":"J. Acoust. Soc. Am."},{"key":"550_CR15","doi-asserted-by":"crossref","unstructured":"Petridis, S., Stafylakis, T., Ma, P., Tzimiropoulos, G. & Pantic, M. Audio-visual speech recognition with a hybrid CTC\/attention architecture. In Proc. IEEE Spoken Language Technology Workshop 513\u2013520 (IEEE, 2018).","DOI":"10.1109\/SLT.2018.8639643"},{"key":"550_CR16","doi-asserted-by":"crossref","unstructured":"Yu, J. et al. Audio-visual recognition of overlapped speech for the LRS2 dataset. In Proc. 45th IEEE International Conference on Acoustics, Speech and Signal Processing 6984\u20136988 (IEEE, 2020).","DOI":"10.1109\/ICASSP40776.2020.9054127"},{"key":"550_CR17","doi-asserted-by":"crossref","unstructured":"Yu, W., Zeiler, S. & Kolossa, D. Fusing information streams in end-to-end audio-visual speech recognition. In Proc. 46th IEEE International Conference on Acoustics, Speech and Signal Processing 3430\u20133434 (IEEE, 2021).","DOI":"10.1109\/ICASSP39728.2021.9414553"},{"key":"550_CR18","first-page":"1052","volume":"28","author":"G Sterpu","year":"2020","unstructured":"Sterpu, G., Saam, C. & Harte, N. How to teach DNNs to pay attention to the visual modality in speech recognition. IEEE\/ACM Trans. Audio Speech Language Process. 28, 1052\u20131064 (2020).","journal-title":"IEEE"},{"key":"550_CR19","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J. S. & Zisserman, A. The conversation: deep audio-visual speech enhancement. In Proc. 19th Annual Conference of International Speech Communication Association 3244\u20133248 (ISCA, 2018).","DOI":"10.21437\/Interspeech.2018-1400"},{"key":"550_CR20","doi-asserted-by":"publisher","first-page":"112:1\u2013112:11","DOI":"10.1145\/3197517.3201357","volume":"37","author":"A Ephrat","year":"2018","unstructured":"Ephrat, A. et al. Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation. ACM Trans. Graph. 37, 112:1\u2013112:11 (2018).","journal-title":"ACM Trans. Graph."},{"key":"550_CR21","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Hayashi, T., Takeda, K. & Watanabe, S. End-to-end automatic speech recognition integrated with CTC-based voice activity detection. In Proc. 45th IEEE International Conference on Acoustics, Speech and Signal Processing 6999\u20137003 (IEEE, 2020).","DOI":"10.1109\/ICASSP40776.2020.9054358"},{"key":"550_CR22","doi-asserted-by":"crossref","unstructured":"Kim, Y. J. et al. Look who\u2019s talking: active speaker detection in the wild. In Proc. 22nd Annual Conference of International Speech Communication Association 3675\u20133679 (ISCA, 2021).","DOI":"10.21437\/Interspeech.2021-2041"},{"key":"550_CR23","doi-asserted-by":"crossref","unstructured":"Chung, J. S., Huh, J., Nagrani, A., Afouras, T. & Zisserman, A. Spot the conversation: speaker diarisation in the wild. In Proc. 21st Annual Conference of International Speech Communication Association 299\u2013303 (ISCA, 2020).","DOI":"10.21437\/Interspeech.2020-2337"},{"key":"550_CR24","doi-asserted-by":"publisher","first-page":"270","DOI":"10.1016\/j.specom.2009.08.002","volume":"52","author":"B Denby","year":"2010","unstructured":"Denby, B. et al. Silent speech interfaces. Speech Commun. 52, 270\u2013287 (2010).","journal-title":"Speech Commun."},{"key":"550_CR25","doi-asserted-by":"crossref","unstructured":"Haliassos, A., Vougioukas, K., Petridis, S. & Pantic, M. Lips don\u2019t lie: a generalisable and robust approach to face forgery detection. In Proc. 34th IEEE\/CVF Conference on Computer Vision and Pattern Recognition 5039\u20135049 (IEEE, 2021).","DOI":"10.1109\/CVPR46437.2021.00500"},{"key":"550_CR26","doi-asserted-by":"crossref","unstructured":"Mira, R. et al. End-to-end video-to-speech synthesis using generative adversarial networks. IEEE Transactions on Cybernetics. 1\u201313 (IEEE, 2022).","DOI":"10.1109\/TCYB.2022.3162495"},{"key":"550_CR27","doi-asserted-by":"crossref","unstructured":"Prajwal, K., Mukhopadhyay, R., Namboodiri, V. P. & Jawahar, C. Learning individual speaking styles for accurate lip to speech synthesis. In Proc. 33rd IEEE\/CVF Conference on Computer Vision and Pattern Recognition 13796\u201313805 (IEEE, 2020).","DOI":"10.1109\/CVPR42600.2020.01381"},{"key":"550_CR28","doi-asserted-by":"crossref","unstructured":"Dungan, L., Karaali, A. & Harte, N. The impact of reduced video quality on visual speech recognition. In Proc. 25th IEEE International Conference on Image Processing 2560\u20132564 (IEEE, 2018).","DOI":"10.1109\/ICIP.2018.8451754"},{"key":"550_CR29","doi-asserted-by":"crossref","unstructured":"Bear, H. L., Harvey, R., Theobald, B.-J. & Lan, Y. Resolution limits on visual speech recognition. In Proc. 21st IEEE International Conference on Image Processing 1371\u20131375 (IEEE, 2014).","DOI":"10.1109\/ICIP.2014.7025274"},{"key":"550_CR30","unstructured":"Geirhos, R. et al. ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness. In Proc. 7th International Conference on Learning Representations (OpenReview, 2019)."},{"key":"550_CR31","doi-asserted-by":"crossref","unstructured":"Cheng, S. et al. Towards pose-invariant lip-reading. In Proc. 45th IEEE International Conference on Acoustics, Speech and Signal Processing 4357\u20134361 (IEEE, 2020).","DOI":"10.1109\/ICASSP40776.2020.9054384"},{"key":"550_CR32","doi-asserted-by":"crossref","unstructured":"Wand, M. & Schmidhuber, J. Improving speaker-independent lipreading with domain-adversarial training. In Proc. 18th Annual Conference of International Speech Communication Association 3662\u20133666 (ISCA, 2017).","DOI":"10.21437\/Interspeech.2017-421"},{"key":"550_CR33","doi-asserted-by":"publisher","unstructured":"Petridis, S., Wang, Y., Li, Z. & Pantic, M. End-to-end multi-view lipreading. In Proc. 28th British Machine Vision Conference (BMVA, 2017); https:\/\/doi.org\/10.5244\/C.31.161","DOI":"10.5244\/C.31.161"},{"key":"550_CR34","first-page":"17\u201324","volume":"44","author":"K Bicevskis","year":"2016","unstructured":"Bicevskis, K. et al. Effects of mouthing and interlocutor presence on movements of visible vs. non-visible articulators. Can. Acoust. 44, 17\u201324 (2016).","journal-title":"Can. Acoust."},{"key":"550_CR35","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1121\/1.4939495","volume":"139","author":"J \u0160imko","year":"2016","unstructured":"\u0160imko, J., Be\u0148u\u0161, \u0160. & Vainio, M. Hyperarticulation in Lombard speech: global coordination of the jaw, lips and the tongue. J. Acoust. Soc. Am. 139, 151\u2013162 (2016).","journal-title":"J. Acoust. Soc. Am."},{"key":"550_CR36","doi-asserted-by":"crossref","unstructured":"Ma, P., Petridis, S. & Pantic, M. Investigating the Lombard effect influence on end-to-end audio-visual speech recognition. In Proc. 20th Annual Conference of International Speech Communication Association 4090\u20134094 (ISCA, 2019).","DOI":"10.21437\/Interspeech.2019-2726"},{"key":"550_CR37","doi-asserted-by":"crossref","unstructured":"Petridis, S., Shen, J., Cetin, D. & Pantic, M. Visual-only recognition of normal, whispered and silent speech. In Proc. 43rd IEEE International Conference on Acoustics, Speech and Signal Processing 6219\u20136223 (IEEE, 2018).","DOI":"10.1109\/ICASSP.2018.8461596"},{"key":"550_CR38","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1016\/j.csl.2012.06.003","volume":"27","author":"P Heracleous","year":"2013","unstructured":"Heracleous, P., Ishi, C. T., Sato, M., Ishiguro, H. & Hagita, N. Analysis of the visual Lombard effect and automatic recognition experiments. Comput. Speech Language 27, 288\u2013300 (2013).","journal-title":"Comput. Speech Language"},{"key":"550_CR39","unstructured":"Efforts to acknowledge the risks of new A.I. technology. New York Times (22 October 2018); https:\/\/www.nytimes.com\/2018\/10\/22\/business\/efforts-to-acknowledge-the-risks-of-new-ai-technology.html"},{"key":"550_CR40","unstructured":"Feathers, T. Tech Companies Are Training AI to Read Your Lips https:\/\/www.vice.com\/en\/article\/bvzvdw\/tech-companies-are-training-ai-to-read-your-lips (2021)."},{"key":"550_CR41","unstructured":"Liopa. https:\/\/liopa.ai. Accessed 24 November 2021."},{"key":"550_CR42","unstructured":"Crawford, S. Facial recognition laws are (literally) all over the map. Wired (16 December 2019); https:\/\/www.wired.com\/story\/facial-recognition-laws-are-literally-all-over-the-map\/"},{"key":"550_CR43","unstructured":"Flynn, S. 13 cities where police are banned from using facial recognition tech. Innovation & Tech Today (18 November 2020); https:\/\/innotechtoday.com\/13-cities-where-police-are-banned-from-using-facial-recognition-tech\/"},{"key":"550_CR44","unstructured":"An update on our use of face recognition. FaceBook (2 November 2021); https:\/\/about.fb.com\/news\/2021\/11\/update-on-use-of-face-recognition\/"},{"key":"550_CR45","unstructured":"Metz, R. Amazon will block police indefinitely from using its facial-recognition software. CNN (18 May 2021); https:\/\/edition.cnn.com\/2021\/05\/18\/tech\/amazon-police-facial-recognition-ban\/index.html"},{"key":"550_CR46","unstructured":"Greene, J. Microsoft won\u2019t sell police its facial-recognition technology, following similar moves by Amazon and IBM. Washington Post (11 June 2020) https:\/\/www.washingtonpost.com\/technology\/2020\/06\/11\/microsoft-facial-recognition"},{"key":"550_CR47","unstructured":"Afouras, T., Chung, J. S. & Zisserman, A. LRS3-TED: a large-scale dataset for visual speech recognition. Preprint at https:\/\/arxiv.org\/abs\/1809.00496 (2018)."},{"key":"550_CR48","unstructured":"Zadeh, A. B. et al. CMU-MOSEAS: a multimodal language dataset for Spanish, Portuguese, German and French. In Proc. 2020 Conference on Empirical Methods in Natural Language Processing 1801\u20131812 (ACL, 2020)."},{"key":"550_CR49","doi-asserted-by":"crossref","unstructured":"Salesky, E. et al. The multilingual TEDx corpus for speech recognition and translation. In Proc. 22nd Annual Conference of International Speech Communication Association 3655\u20133659 (ISCA, 2021).","DOI":"10.21437\/Interspeech.2021-11"},{"key":"550_CR50","doi-asserted-by":"crossref","unstructured":"Valk, J. & Alum\u00e4e, T. VoxLingua107: a dataset for spoken language recognition. In Proc. IEEE Spoken Language Technology Workshop 652\u2013658 (IEEE, 2021).","DOI":"10.1109\/SLT48900.2021.9383459"},{"key":"550_CR51","doi-asserted-by":"crossref","unstructured":"Deng, J. et al. RetinaFace: single-stage dense face localisation in the wild. In Proc. 33rd IEEE\/CVF Conference on Computer Vision and Pattern Recognition 5203\u20135212 (IEEE, 2020).","DOI":"10.1109\/CVPR42600.2020.00525"},{"key":"550_CR52","doi-asserted-by":"crossref","unstructured":"Bulat, A. & Tzimiropoulos, G. How far are we from solving the 2D & 3D face alignment problem? (and a dataset of 230,000 3D facial landmarks). In Proc. 16th IEEE\/CVF International Conference on Computer Vision 1021\u20131030 (IEEE, 2017).","DOI":"10.1109\/ICCV.2017.116"},{"key":"550_CR53","unstructured":"Simonyan, K. & Zisserman, A. Very deep convolutional networks for large-scale image recognition. In Proc. 3rd International Conference on Learning Representations (OpenReview, 2015)."},{"key":"550_CR54","unstructured":"Assael, Y., Shillingford, B., Whiteson, S. & De Freitas, N. LipNet: end-to-end sentence-level lipreading. Preprint at https:\/\/arxiv.org\/abs\/1611.01599 (2016)."},{"key":"550_CR55","doi-asserted-by":"crossref","unstructured":"Ma, P., Martinez, B., Petridis, S. & Pantic, M. Towards practical lipreading with distilled and efficient models. In Proc. 46th IEEE International Conference on Acoustics, Speech and Signal Processing 7608\u20137612 (IEEE, 2021).","DOI":"10.1109\/ICASSP39728.2021.9415063"},{"key":"550_CR56","doi-asserted-by":"crossref","unstructured":"Park, D. S. et al. SpecAugment: a simple data augmentation method for automatic speech recognition. In Proc. 20th Annual Conference of International Speech Communication Association 2613\u20132617 (ISCA, 2019).","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"550_CR57","doi-asserted-by":"crossref","unstructured":"Liu, C. et al. Improving RNN transducer based ASR with auxiliary tasks. In Proc. IEEE Spoken Language Technology Workshop 172\u2013179 (IEEE, 2021).","DOI":"10.1109\/SLT48900.2021.9383548"},{"key":"550_CR58","doi-asserted-by":"crossref","unstructured":"Toshniwal, S., Tang, H., Lu, L. & Livescu, K. Multitask learning with low-level auxiliary tasks for encoder-decoder based speech recognition. In Proc. 18th Annual Conference of International Speech Communication Association 3532\u20133536 (ISCA, 2017).","DOI":"10.21437\/Interspeech.2017-1118"},{"key":"550_CR59","doi-asserted-by":"crossref","unstructured":"Lee, J. & Watanabe, S. Intermediate loss regularization for CTC-based speech recognition. In Proc. 46th IEEE International Conference on Acoustics, Speech and Signal Processing 6224\u20136228 (IEEE, 2021).","DOI":"10.1109\/ICASSP39728.2021.9414594"},{"key":"550_CR60","doi-asserted-by":"crossref","unstructured":"Pascual, S., Ravanelli, M., Serr\u00e0, J., Bonafonte, A. & Bengio, Y. Learning problem-agnostic speech representations from multiple self-supervised tasks. In Proc. 20th Annual Conference of International Speech Communication Association 161\u2013165 (ISCA, 2019).","DOI":"10.21437\/Interspeech.2019-2605"},{"key":"550_CR61","unstructured":"Shukla, A., Petridis, S. & Pantic, M. Learning speech representations from raw audio by joint audiovisual self-supervision. In Proc. 37th International Conference on Machine Learning Workshop (PMLR, 2020)."},{"key":"550_CR62","doi-asserted-by":"crossref","unstructured":"Ma, P., Mira, R., Petridis, S., Schuller, B. W. & Pantic, M. LiRA: learning visual speech representations from audio through self-supervision. In Proc. 22nd Annual Conference of International Speech Communication Association 3011\u20133015 (ISCA, 2021).","DOI":"10.21437\/Interspeech.2021-1360"},{"key":"550_CR63","doi-asserted-by":"crossref","unstructured":"Serdyuk, D., Braga, O. & Siohan, O. Transformer-Based Video Front-Ends for Audio-Visual Speech Recognition for Single and Muti-Person Video. In Proc. 23rd Annual Conference of International Speech Communication Association 2833\u20132837 (ISCA, 2022).","DOI":"10.21437\/Interspeech.2022-10920"},{"key":"550_CR64","doi-asserted-by":"crossref","unstructured":"Watanabe, S. et al. ESPnet: End-to-end speech processing toolkit. In Proc. 19th Annual Conference of International Speech Communication Association 2207\u20132211 (ISCA, 2018).","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"550_CR65","unstructured":"Kingma, D. & Ba, J. Adam: a method for stochastic optimization. In Proc. 2nd International Conference on Learning Representations (OpenReview, 2014)."},{"key":"550_CR66","doi-asserted-by":"publisher","unstructured":"Ma, P., Petridis, S. & Pantic, M. 2022. mpc001\/Visual_Speech_Recognition_for_Multiple_Languages: visual speech recognition for multiple languages. Zenodo https:\/\/doi.org\/10.5281\/zenodo.7065080","DOI":"10.5281\/zenodo.7065080"},{"key":"550_CR67","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J. S. & Zisserman, A. ASR is all you need: cross-modal distillation for lip reading. In Proc. 45th IEEE International Conference on Acoustics, Speech and Signal Processing 2143\u20132147 (IEEE, 2020).","DOI":"10.1109\/ICASSP40776.2020.9054253"},{"key":"550_CR68","doi-asserted-by":"crossref","unstructured":"Ren, S., Du, Y., Lv, J., Han, G. & He, S. Learning from the master: distilling cross-modal advanced knowledge for lip reading. In Proc. 34th IEEE\/CVF Conference on Computer Vision and Pattern Recognition 13325\u201313333 (IEEE, 2021).","DOI":"10.1109\/CVPR46437.2021.01312"},{"key":"550_CR69","doi-asserted-by":"crossref","unstructured":"Zhao, Y. et al. Hearing lips: improving lip reading by distilling speech recognizers. In Proc. 34th AAAI Conference on Artificial Intelligence 6917\u20136924 (AAAI, 2020).","DOI":"10.1609\/aaai.v34i04.6174"}],"container-title":["Nature Machine Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.nature.com\/articles\/s42256-022-00550-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-022-00550-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.nature.com\/articles\/s42256-022-00550-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T08:48:08Z","timestamp":1728204488000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.nature.com\/articles\/s42256-022-00550-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,24]]},"references-count":69,"journal-issue":{"issue":"11","published-online":{"date-parts":[[2022,11]]}},"alternative-id":["550"],"URL":"https:\/\/doi.org\/10.1038\/s42256-022-00550-z","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-1386805\/v1","asserted-by":"object"}]},"ISSN":["2522-5839"],"issn-type":[{"value":"2522-5839","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10,24]]},"assertion":[{"value":"22 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 September 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 October 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}