{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:11:29Z","timestamp":1784203889856,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,12]],"date-time":"2024-04-12T00:00:00Z","timestamp":1712880000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,12]]},"DOI":"10.1145\/3661725.3661767","type":"proceedings-article","created":{"date-parts":[[2024,6,20]],"date-time":"2024-06-20T06:22:44Z","timestamp":1718864564000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Speaker Adaptation for Lip Reading with Robust Entropy Minimization and Adaptive Pseudo Labels"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8729-4551","authenticated-orcid":false,"given":"Hao","family":"Shen","sequence":"first","affiliation":[{"name":"State Grid Zhejiang Electric Power Co., Ltd, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8093-4437","authenticated-orcid":false,"given":"Weihao","family":"Jiang","sequence":"additional","affiliation":[{"name":"Zhejiang University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2663-0848","authenticated-orcid":false,"given":"Junjie","family":"Huang","sequence":"additional","affiliation":[{"name":"Zhejiang University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,6,20]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Triantafyllos Afouras Joon\u00a0Son Chung and Andrew Zisserman. 2020. ASR is all you need: cross-modal distillation for lip reading. arxiv:1911.12747\u00a0[cs.CV]","DOI":"10.1109\/ICASSP40776.2020.9054253"},{"key":"e_1_3_2_1_3_1","unstructured":"Yannis\u00a0M. Assael Brendan Shillingford Shimon Whiteson and Nando de Freitas. 2016. LipNet: End-to-End Sentence-level Lipreading. arxiv:1611.01599\u00a0[cs.LG]"},{"key":"e_1_3_2_1_4_1","unstructured":"Alexei Baevski Arun Babu Wei-Ning Hsu and Michael Auli. 2023. Efficient Self-supervised Learning with Contextualized Target Representations for Vision Speech and Language. arxiv:2212.07525\u00a0[cs.LG]"},{"key":"e_1_3_2_1_5_1","unstructured":"Alexei Baevski Wei-Ning Hsu Qiantong Xu Arun Babu Jiatao Gu and Michael Auli. 2022. data2vec: A General Framework for Self-supervised Learning in Speech Vision and Language. arxiv:2202.03555\u00a0[cs.LG]"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413623"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.2229005"},{"key":"e_1_3_2_1_8_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805\u00a0[cs.CL]","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv:1810.04805\u00a0[cs.CL]"},{"key":"e_1_3_2_1_9_1","volume-title":"NeurIPS DistShift Workshop.","author":"Fleuret Fran\u00e7ois","year":"2021","unstructured":"Fran\u00e7ois Fleuret 2021. Test time adaptation through perturbation robustness. In NeurIPS DistShift Workshop."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01134"},{"key":"e_1_3_2_1_11_1","unstructured":"Sachin Goyal Mingjie Sun Aditi Raghunathan and Zico Kolter. 2022. Test-Time Adaptation via Conjugate Pseudo-labels. arxiv:2207.09640\u00a0[cs.LG]"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Pujitha\u00a0Appan Kandala Abhinav Thanda Dilip\u00a0Kumar Margam Rohith\u00a0Chandrashekar Aralikatti Tanay Sharma Sharad Roy and Shankar\u00a0M. Venkatesan. 2019. Speaker Adaptation for Lip-Reading Using Visual Identity Vectors. In Interspeech. https:\/\/api.semanticscholar.org\/CorpusID:202705816","DOI":"10.21437\/Interspeech.2019-3237"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2645"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Sameer Khurana Niko Moritz Takaaki Hori and Jonathan Le\u00a0Roux. 2021. Unsupervised domain adaptation for speech recognition via uncertainty driven self-training. In ICASSP.","DOI":"10.1109\/ICASSP39728.2021.9414299"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Ond\u0159ej Klejch Joachim Fainberg Peter Bell and Steve Renals. 2019. Speaker Adaptive Training using Model Agnostic Meta-Learning. arxiv:1910.10605\u00a0[cs.CL]","DOI":"10.1109\/ASRU46091.2019.9003751"},{"key":"e_1_3_2_1_17_1","unstructured":"Jogendra\u00a0Nath Kundu Akshay Kulkarni Suvaansh Bhambri Deepesh Mehta Shreyas Kulkarni Varun Jampani and R.\u00a0Venkatesh Babu. 2022. Balancing Discriminability and Transferability for Source-Free Domain Adaptation. arxiv:2206.08009\u00a0[cs.CV]"},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 4544\u20134553","author":"Kundu Jogendra\u00a0Nath","year":"2020","unstructured":"Jogendra\u00a0Nath Kundu, Naveen Venkat, R\u00a0Venkatesh Babu, 2020. Universal source-free domain adaptation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 4544\u20134553."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Bo Li and Khe\u00a0Chai Sim. 2010. Comparison of discriminative input and output transformations for speaker adaptation in the hybrid NN\/HMM systems. In Interspeech. https:\/\/api.semanticscholar.org\/CorpusID:9315703","DOI":"10.21437\/Interspeech.2010-214"},{"key":"e_1_3_2_1_20_1","unstructured":"Jian Liang Dapeng Hu and Jiashi Feng. 2021. Do We Really Need to Access the Source Data? Source Hypothesis Transfer for Unsupervised Domain Adaptation. arxiv:2002.08546\u00a0[cs.CV]"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639212"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475220"},{"key":"e_1_3_2_1_23_1","volume-title":"When does self-supervised test-time training fail or thrive?Advances in Neural Information Processing Systems 34","author":"Liu Yuejiang","year":"2021","unstructured":"Yuejiang Liu, Parth Kothari, Bastien Van\u00a0Delft, Baptiste Bellot-Gurlet, Taylor Mordan, and Alexandre Alahi. 2021. Ttt++: When does self-supervised test-time training fail or thrive?Advances in Neural Information Processing Systems 34 (2021), 21808\u201321820."},{"key":"e_1_3_2_1_24_1","volume-title":"Advances in Neural Information Processing Systems, M.\u00a0Ranzato, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.S. Liang, and J.\u00a0Wortman Vaughan (Eds.). Vol.\u00a034. Curran Associates","author":"Liu Yuejiang","year":"1808","unstructured":"Yuejiang Liu, Parth Kothari, Bastien van Delft, Baptiste Bellot-Gurlet, Taylor Mordan, and Alexandre Alahi. 2021. TTT++: When Does Self-Supervised Test-Time Training Fail or Thrive?. In Advances in Neural Information Processing Systems, M.\u00a0Ranzato, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.S. Liang, and J.\u00a0Wortman Vaughan (Eds.). Vol.\u00a034. Curran Associates, Inc., 21808\u201321820. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2021\/file\/b618c3210e934362ac261db280128c22-Paper.pdf"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.2019.8682510"},{"key":"e_1_3_2_1_26_1","volume-title":"Towards stable test-time adaptation in dynamic wild world. arXiv preprint arXiv:2302.12400","author":"Niu Shuaicheng","year":"2023","unstructured":"Shuaicheng Niu, Jiaxiang Wu, Yifan Zhang, Zhiquan Wen, Yaofo Chen, Peilin Zhao, and Mingkui Tan. 2023. Towards stable test-time adaptation in dynamic wild world. arXiv preprint arXiv:2302.12400 (2023)."},{"key":"e_1_3_2_1_27_1","unstructured":"Marzieh Oghbaie Arian Sabaghi Kooshan Hashemifard and Mohammad Akbari. 2021. Advances and Challenges in Deep Lip Reading. arxiv:2110.07879\u00a0[cs.CV]"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639643"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01312"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01312"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Themos Stafylakis and Georgios Tzimiropoulos. 2017. Combining Residual Networks with LSTMs for Lipreading. arxiv:1703.04105\u00a0[cs.CV]","DOI":"10.21437\/Interspeech.2017-85"},{"key":"e_1_3_2_1_35_1","volume-title":"International conference on machine learning. PMLR, 9229\u20139248","author":"Sun Yu","year":"2020","unstructured":"Yu Sun, Xiaolong Wang, Zhuang Liu, John Miller, Alexei Efros, and Moritz Hardt. 2020. Test-time training with self-supervision for generalization under distribution shifts. In International conference on machine learning. PMLR, 9229\u20139248."},{"key":"e_1_3_2_1_36_1","volume-title":"Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. Advances in neural information processing systems 30","author":"Tarvainen Antti","year":"2017","unstructured":"Antti Tarvainen and Harri Valpola. 2017. Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"e_1_3_2_1_38_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_39_1","volume-title":"Tent: Fully test-time adaptation by entropy minimization. arXiv preprint arXiv:2006.10726","author":"Wang Dequan","year":"2020","unstructured":"Dequan Wang, Evan Shelhamer, Shaoteng Liu, Bruno Olshausen, and Trevor Darrell. 2020. Tent: Fully test-time adaptation by entropy minimization. arXiv preprint arXiv:2006.10726 (2020)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00706"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP40778.2020.9190780"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639201"},{"key":"e_1_3_2_1_43_1","volume-title":"MEMO: Test Time Robustness via Adaptation and Augmentation. arXiv preprint arXiv:2110.09506","author":"Zhang Marvin","year":"2021","unstructured":"Marvin Zhang, Sergey Levine, and Chelsea Finn. 2021. MEMO: Test Time Robustness via Adaptation and Augmentation. arXiv preprint arXiv:2110.09506 (2021)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP42928.2021.9506396"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00080"},{"key":"e_1_3_2_1_46_1","volume-title":"International Conference on Machine Learning. PMLR, 42574\u201342588","author":"Zhou Zhi","year":"2023","unstructured":"Zhi Zhou, Lan-Zhe Guo, Lin-Han Jia, Dingchu Zhang, and Yu-Feng Li. 2023. ODS: test-time adaptation in the presence of open-world data shift. In International Conference on Machine Learning. PMLR, 42574\u201342588."}],"event":{"name":"CMLDS 2024: 2024 International Conference on Computing, Machine Learning and Data Science","location":"Singapore Singapore","acronym":"CMLDS 2024"},"container-title":["International Conference on Computing Machine Learning and Data Science"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3661725.3661767","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3661725.3661767","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T15:06:37Z","timestamp":1756825597000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3661725.3661767"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,12]]},"references-count":46,"alternative-id":["10.1145\/3661725.3661767","10.1145\/3661725"],"URL":"https:\/\/doi.org\/10.1145\/3661725.3661767","relation":{},"subject":[],"published":{"date-parts":[[2024,4,12]]},"assertion":[{"value":"2024-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}