{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:58:50Z","timestamp":1782835130243,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,4,19]],"date-time":"2023-04-19T00:00:00Z","timestamp":1681862400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"JST CREST","award":["JPMJCR17A3"],"award-info":[{"award-number":["JPMJCR17A3"]}]},{"name":"JST Moonshot R&D","award":["JPMJMS2012"],"award-info":[{"award-number":["JPMJMS2012"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,4,19]]},"DOI":"10.1145\/3544548.3581465","type":"proceedings-article","created":{"date-parts":[[2023,4,20]],"date-time":"2023-04-20T04:27:55Z","timestamp":1681964875000},"page":"1-21","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":38,"title":["LipLearner: Customizable Silent Speech Interactions on Mobile Devices"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6048-3268","authenticated-orcid":false,"given":"Zixiong","family":"Su","sequence":"first","affiliation":[{"name":"Rekimoto Lab, GSII, The University of Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1401-8482","authenticated-orcid":false,"given":"Shitao","family":"Fang","sequence":"additional","affiliation":[{"name":"Interactive Intelligent Systems Laboratory, The University of Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3629-2514","authenticated-orcid":false,"given":"Jun","family":"Rekimoto","sequence":"additional","affiliation":[{"name":"The University of Tokyo, Japan and Sony CSL Kyoto, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,4,19]]},"reference":[{"key":"e_1_3_3_2_1_1","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2552\/ab0c59"},{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Aaron Bangor Philip\u00a0T Kortum and James\u00a0T Miller. 2008. An empirical evaluation of the system usability scale. Intl. Journal of Human\u2013Computer Interaction 24 6(2008) 574\u2013594.","DOI":"10.1080\/10447310802205776"},{"key":"e_1_3_3_2_4_1","volume-title":"SUS-A quick and dirty usability scale. Usability evaluation in industry 189, 194","author":"John Brooke","year":"1996","unstructured":"John Brooke 1996. SUS-A quick and dirty usability scale. Usability evaluation in industry 189, 194 (1996), 4\u20137."},{"key":"e_1_3_3_2_5_1","volume-title":"Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.F. Balcan, and H.\u00a0Lin (Eds.). Vol.\u00a033. Curran Associates","author":"Brown Tom","year":"1877","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.F. Balcan, and H.\u00a0Lin (Eds.). Vol.\u00a033. Curran Associates, Inc., 1877\u20131901. https:\/\/proceedings.neurips.cc\/paper\/2020\/file\/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3379337.3415879"},{"key":"e_1_3_3_2_7_1","unstructured":"Wei-Yu Chen Yen-Cheng Liu Zsolt Kira Yu-Chiang\u00a0Frank Wang and Jia-Bin Huang. 2019. A Closer Look at Few-shot Classification. CoRR abs\/1904.04232(2019). arXiv:1904.04232http:\/\/arxiv.org\/abs\/1904.04232"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"e_1_3_3_2_9_1","volume-title":"Asian conference on computer vision. Springer, 87\u2013103","author":"Chung Joon\u00a0Son","year":"2016","unstructured":"Joon\u00a0Son Chung and Andrew Zisserman. 2016. Lip reading in the wild. In Asian conference on computer vision. Springer, 87\u2013103."},{"key":"e_1_3_3_2_10_1","volume-title":"Retrieved","author":"Statista\u00a0Research Department","year":"2022","unstructured":"Statista\u00a0Research Department. 2022. Main devices used with voice assistants in the U.S. 2021, by brand. Retrieved April 28, 2022 from https:\/\/www.statista.com\/statistics\/1274398\/voice-assistant-use-by-device-united-states\/"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CA.1998.681913"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1008166717597"},{"key":"e_1_3_3_2_13_1","volume-title":"Development of a (silent) speech recognition system for patients following laryngectomy. Medical engineering & physics 30, 4","author":"Fagan J","year":"2008","unstructured":"Michael\u00a0J Fagan, Stephen\u00a0R Ell, James\u00a0M Gilbert, E Sarrazin, and Peter\u00a0M Chapman. 2008. Development of a (silent) speech recognition system for patients following laryngectomy. Medical engineering & physics 30, 4 (2008), 419\u2013425."},{"key":"e_1_3_3_2_14_1","unstructured":"Dalu Feng Shuang Yang Shiguang Shan and Xilin Chen. 2020. Learn an effective lip reading model without pains. arXiv preprint arXiv:2011.07557(2020)."},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242603"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01524"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2016.02.002"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0008218"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3463506"},{"key":"e_1_3_3_2_20_1","volume-title":"Brain-to-text: decoding spoken phrases from phone representations in the brain. Frontiers in neuroscience 9","author":"Herff Christian","year":"2015","unstructured":"Christian Herff, Dominic Heger, Adriana De\u00a0Pesters, Dominic Telaar, Peter Brunner, Gerwin Schalk, and Tanja Schultz. 2015. Brain-to-text: decoding spoken phrases from phone representations in the brain. Frontiers in neuroscience 9 (2015), 217."},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475420"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.11.004"},{"key":"e_1_3_3_2_23_1","unstructured":"Apple Inc.2022. Core ML | Apple Developer Documentation. Retrieved Feb. 9 2023 from https:\/\/developer.apple.com\/documentation\/coreml"},{"key":"e_1_3_3_2_24_1","unstructured":"Apple Inc.2022. Create ML | Apple Developer Documentation. Retrieved Feb. 9 2023 from https:\/\/developer.apple.com\/documentation\/createml"},{"key":"e_1_3_3_2_25_1","unstructured":"Apple Inc.2022. SFSpeechRecognizer | Apple Developer Documentation. Retrieved Feb. 9 2023 from https:\/\/developer.apple.com\/documentation\/speech\/sfspeechrecognizer"},{"key":"e_1_3_3_2_26_1","volume-title":"Shortcuts User Guide - Apple Support. Retrieved","author":"Apple Inc.","year":"2023","unstructured":"Apple Inc.2022. Shortcuts User Guide - Apple Support. Retrieved Feb. 9, 2023 from https:\/\/support.apple.com\/guide\/shortcuts\/welcome\/ios"},{"key":"e_1_3_3_2_27_1","volume-title":"Vision | Apple Developer Documentation. Retrieved","author":"Apple Inc.","year":"2023","unstructured":"Apple Inc.2022. Vision | Apple Developer Documentation. Retrieved Feb. 9, 2023 from https:\/\/developer.apple.com\/documentation\/vision"},{"key":"e_1_3_3_2_28_1","volume-title":"What can I ask Siri? - Official Apple Support. Retrieved","author":"Apple Inc.","year":"2023","unstructured":"Apple Inc.2022. What can I ask Siri? - Official Apple Support. Retrieved Feb. 9, 2023 from https:\/\/support.apple.com\/siri"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502020"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2018.02.002"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3172944.3172977"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"crossref","unstructured":"Vahid Kazemi and Josephine Sullivan. 2014. One Millisecond Face Alignment with an Ensemble of Regression Trees. In CVPR.","DOI":"10.1109\/CVPR.2014.241"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502015"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3399715.3399852"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300376"},{"key":"e_1_3_3_2_36_1","unstructured":"Naoki Kimura Zixiong Su and Takaaki Saeki. 2020. End-to-End Deep Learning Speech Recognition Model for Silent Speech Challenge.. In INTERSPEECH. 1025\u20131026."},{"key":"e_1_3_3_2_37_1","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference. 6866\u20136873","author":"Kimura Naoki","year":"2022","unstructured":"Naoki Kimura, Zixiong Su, Takaaki Saeki, and Jun Rekimoto. 2022. SSR7000: A Synchronized Corpus of Ultrasound Tongue Imaging for End-to-End Silent Speech Recognition. In Proceedings of the Thirteenth Language Resources and Evaluation Conference. 6866\u20136873."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/1577069.1755843"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45683-X_60"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3311823.3311831"},{"key":"e_1_3_3_2_41_1","volume-title":"Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172(2019).","author":"Lugaresi Camillo","year":"2019","unstructured":"Camillo Lugaresi, Jiuqiang Tang, Hadon Nash, Chris McClanahan, Esha Uboweja, Michael Hays, Fan Zhang, Chuo-Ling Chang, Ming\u00a0Guang Yong, Juhyun Lee, 2019. Mediapipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172(2019)."},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053841"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1088\/1741-2552\/aac965"},{"key":"e_1_3_3_2_44_1","volume-title":"Voice Assistant Anyone? Yes please, but not in public. Creative Strategies","author":"Milanesi Carolina","year":"2016","unstructured":"Carolina Milanesi. 2016. Voice Assistant Anyone? Yes please, but not in public. Creative Strategies (2016)."},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445565"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461326"},{"key":"e_1_3_3_2_47_1","unstructured":"Anne Porbadnigk Marek Wester Jan-P Calliess and Tanja Schultz. 2009. EEG-based speech recognition."},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01381"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13311-018-00692-2"},{"key":"e_1_3_3_2_50_1","volume-title":"International Conference on Machine Learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_51_1","first-page":"1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter\u00a0J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. Journal of Machine Learning Research 21, 140 (2020), 1\u201367. http:\/\/jmlr.org\/papers\/v21\/20-074.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.21437\/AVSP.2019-17"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1016\/0093-934X(87)90058-7"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3196709.3196772"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00458"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475415"},{"key":"e_1_3_3_2_57_1","volume-title":"Precise and Expressive Interactions Combining Gaze Input and Silent Speech Commands for Hands-free Smart TV Control. In ACM Symposium on Eye Tracking Research and Applications. 1\u20136.","author":"Su Zixiong","year":"2021","unstructured":"Zixiong Su, Xinlei Zhang, Naoki Kimura, and Jun Rekimoto. 2021. Gaze+ Lip: Rapid, Precise and Expressive Interactions Combining Gaze Input and Silent Speech Commands for Hands-free Smart TV Control. In ACM Symposium on Eye Tracking Research and Applications. 1\u20136."},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242599"},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2012.2205241"},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2009.4960405"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"crossref","unstructured":"Tomoki Toda and Kiyohiro Shikano. 2005. NAM-to-speech conversion with Gaussian mixture models. (2005).","DOI":"10.21437\/Interspeech.2005-611"},{"key":"e_1_3_3_2_62_1","volume-title":"The CMU Pronouncing Dictionary. Retrieved","author":"Carnegie\u00a0Mellon University","year":"2023","unstructured":"Carnegie\u00a0Mellon University. 2011. The CMU Pronouncing Dictionary. Retrieved Feb. 9, 2023 from http:\/\/www.speech.cs.cmu.edu\/cgi-bin\/cmudict"},{"key":"e_1_3_3_2_63_1","volume-title":"Representation learning with contrastive predictive coding. arXiv e-prints","author":"Oord Aaron Van\u00a0den","year":"2018","unstructured":"Aaron Van\u00a0den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv e-prints (2018), arXiv\u20131807."},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-48239-3_65"},{"key":"e_1_3_3_2_65_1","unstructured":"Michael Wand and Tanja Schultz. 2011. Session-independent EMG-based Speech Recognition.. In Biosignals. 295\u2013300."},{"key":"e_1_3_3_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747427"},{"key":"e_1_3_3_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376875"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501904"},{"key":"e_1_3_3_2_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/3494987"}],"event":{"name":"CHI '23: CHI Conference on Human Factors in Computing Systems","location":"Hamburg Germany","acronym":"CHI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3581465","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3544548.3581465","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:46:56Z","timestamp":1750178816000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3581465"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,19]]},"references-count":69,"alternative-id":["10.1145\/3544548.3581465","10.1145\/3544548"],"URL":"https:\/\/doi.org\/10.1145\/3544548.3581465","relation":{},"subject":[],"published":{"date-parts":[[2023,4,19]]},"assertion":[{"value":"2023-04-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}