{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T12:55:07Z","timestamp":1782305707892,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","funder":[{"name":"JST Moonshot R&D","award":["JPMJMS2012"],"award-info":[{"award-number":["JPMJMS2012"]}]},{"name":"JSPS KAKENHI","award":["24KJ0775"],"award-info":[{"award-number":["24KJ0775"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,8]]},"DOI":"10.1145\/3719160.3736612","type":"proceedings-article","created":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T10:27:39Z","timestamp":1751711259000},"page":"1-14","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Multimodal Silent Speech-based Text Entry with Word-initials Conditioned LLM"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6048-3268","authenticated-orcid":false,"given":"Zixiong","family":"Su","sequence":"first","affiliation":[{"name":"The University of Tokyo, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1401-8482","authenticated-orcid":false,"given":"Shitao","family":"Fang","sequence":"additional","affiliation":[{"name":"Interactive Intelligent Systems Laboratory, The University of Tokyo, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3629-2514","authenticated-orcid":false,"given":"Jun","family":"Rekimoto","sequence":"additional","affiliation":[{"name":"Sony CSL Kyoto, Kyoto, Japan and The University of Tokyo, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,7]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Triantafyllos Afouras Joon\u00a0Son Chung Andrew Senior Oriol Vinyals and Andrew Zisserman. 2018. Deep audio-visual speech recognition. IEEE transactions on pattern analysis and machine intelligence 44 12 (2018) 8717\u20138727.","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"e_1_3_3_2_3_2","unstructured":"Triantafyllos Afouras Joon\u00a0Son Chung and Andrew Zisserman. 2018. LRS3-TED: a large-scale dataset for visual speech recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1809.00496 (2018)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Alan\u00a0D Baddeley Susan\u00a0E Gathercole and Costanza Papagno. 2017. The phonological loop as a language learning device. Exploring working memory (2017) 164\u2013198.","DOI":"10.4324\/9781315111261-14"},{"key":"e_1_3_3_2_5_2","first-page":"168","volume-title":"Graphics Interface","author":"Bellman Tom","year":"1998","unstructured":"Tom Bellman and I\u00a0Scott MacKenzie. 1998. A probabilistic character layout strategy for mobile text entry. In Graphics Interface, Vol.\u00a098. 168\u2013176."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1145\/2380116.2380136"},{"key":"e_1_3_3_2_7_2","unstructured":"Vladimir\u00a0V Bochkarev Anna\u00a0V Shevlyakova and Valery\u00a0D Solovyev. 2015. The average word length dynamics as an indicator of cultural changes in society. Social Evolution and History 14 2 (2015) 153\u2013175."},{"key":"e_1_3_3_2_8_2","unstructured":"PyTorch Contributors. 2024. CosineAnnealingLR. https:\/\/pytorch.org\/docs\/stable\/generated\/torch.optim.lr_scheduler.CosineAnnealingLR.html. Accessed: 2025-02-27."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3379337.3415857"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00525"},{"key":"e_1_3_3_2_11_2","unstructured":"P\u00a0Kingma Diederik. 2014. Adam: A method for stochastic optimization. (No Title) (2014)."},{"key":"e_1_3_3_2_12_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et\u00a0al. 2024. The llama 3 herd of models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Michael\u00a0J Fagan Stephen\u00a0R Ell James\u00a0M Gilbert E Sarrazin and Peter\u00a0M Chapman. 2008. Development of a (silent) speech recognition system for patients following laryngectomy. Medical engineering & physics 30 4 (2008) 419\u2013425.","DOI":"10.1016\/j.medengphy.2007.05.003"},{"key":"e_1_3_3_2_14_2","unstructured":"Cathy\u00a0Mengying Fang Phoebe Chua Samantha Chan Joanne Leong Andria Bao and Pattie Maes. 2024. Leveraging AI-Generated Emotional Self-Voice to Nudge People towards their Ideal Selves. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.11531 (2024)."},{"key":"e_1_3_3_2_15_2","unstructured":"Elias Frantar Saleh Ashkboos Torsten Hoefler and Dan Alistarh. 2022. Gptq: Accurate post-training quantization for generative pre-trained transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.17323 (2022)."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"crossref","unstructured":"Jose\u00a0A Gonzalez Lam\u00a0A Cheah James\u00a0M Gilbert Jie Bai Stephen\u00a0R Ell Phil\u00a0D Green and Roger\u00a0K Moore. 2016. A silent speech system based on permanent magnet articulography and direct synthesis. Computer Speech & Language 39 (2016) 67\u201387.","DOI":"10.1016\/j.csl.2016.02.002"},{"key":"e_1_3_3_2_17_2","unstructured":"Robbie Hanson. n. d. CocoaAsyncSocket. https:\/\/github.com\/robbiehanson\/CocoaAsyncSocket. Accessed: February 5 2025."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445501"},{"key":"e_1_3_3_2_19_2","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.09685 (2021)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Thomas Hueber Elie-Laurent Benaroya G\u00e9rard Chollet Bruce Denby G\u00e9rard Dreyfus and Maureen Stone. 2010. Development of a silent speech interface driven by ultrasound and optical images of the tongue and lips. Speech Communication 52 4 (2010) 288\u2013300.","DOI":"10.1016\/j.specom.2009.11.004"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3405755.3406130"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3172944.3172977"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502015"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300376"},{"key":"e_1_3_3_2_25_2","first-page":"1025","volume-title":"INTERSPEECH","author":"Kimura Naoki","year":"2020","unstructured":"Naoki Kimura, Zixiong Su, and Takaaki Saeki. 2020. End-to-End Deep Learning Speech Recognition Model for Silent Speech Challenge.. In INTERSPEECH. 1025\u20131026."},{"key":"e_1_3_3_2_26_2","first-page":"6866","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference","author":"Kimura Naoki","year":"2022","unstructured":"Naoki Kimura, Zixiong Su, Takaaki Saeki, and Jun Rekimoto. 2022. Ssr7000: A synchronized corpus of ultrasound tongue imaging for end-to-end silent speech recognition. In Proceedings of the Thirteenth Language Resources and Evaluation Conference. 6866\u20136873."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376317"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Tianshi Li Philip Quinn and Shumin Zhai. 2023. C-PAK: correcting and completing variable-length prefix-based abbreviated keystrokes. ACM Transactions on Computer-Human Interaction 30 1 (2023) 1\u201335.","DOI":"10.1145\/3544101"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096889"},{"key":"e_1_3_3_2_30_2","unstructured":"Ziyang Ma Guanrou Yang Yifan Yang Zhifu Gao Jiaming Wang Zhihao Du Fan Yu Qian Chen Siqi Zheng Shiliang Zhang et\u00a0al. 2024. An Embarrassingly Simple Approach for LLM with Strong ASR Capacity. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.08846 (2024)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45756-9_16"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"I\u00a0Scott MacKenzie and R\u00a0William Soukoreff. 2002. Text entry for mobile computing: Models and methods theory and practice. Human\u2013Computer Interaction 17 2-3 (2002) 147\u2013198.","DOI":"10.1207\/S15327051HCI172&3_2"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/302979.302983"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3571884.3597134"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3571884.3597130"},{"key":"e_1_3_3_2_36_2","unstructured":"Tom Ouyang David Rybach Fran\u00e7oise Beaufays and Michael Riley. 2017. Mobile keyboard input decoding with finite-state transducers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1704.03987 (2017)."},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676401"},{"key":"e_1_3_3_2_38_2","unstructured":"Anne Porbadnigk Marek Wester Jan-P Calliess and Tanja Schultz. 2009. EEG-based speech recognition."},{"key":"e_1_3_3_2_39_2","first-page":"28492","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492\u201328518."},{"key":"e_1_3_3_2_40_2","unstructured":"Sherry Ruan Jacob\u00a0O Wobbrock Kenny Liou Andrew Ng and James Landay. 2016. Speech is 3x faster than typing for english and mandarin text entry on mobile devices. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1608.07323 (2016)."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Paul\u00a0W Sch\u00f6nle Klaus Gr\u00e4be Peter Wenig J\u00f6rg H\u00f6hne J\u00f6rg Schrader and Bastian Conrad. 1987. Electromagnetic articulography: Use of alternating magnetic fields for tracking movements of multiple points inside and outside the vocal tract. Brain and Language 31 1 (1987) 26\u201335.","DOI":"10.1016\/0093-934X(87)90058-7"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2010.5700825"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/2388676.2388793"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3526114.3558737"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581465"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3448018.3458011"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"crossref","unstructured":"Bernhard Suhm Brad Myers and Alex Waibel. 2001. Multimodal error correction for speech user interfaces. ACM transactions on computer-human interaction (TOCHI) 8 1 (2001) 60\u201398.","DOI":"10.1145\/371127.371166"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242599"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"crossref","unstructured":"Tomoki Toda Mikihiro Nakagiri and Kiyohiro Shikano. 2012. Statistical voice conversion techniques for body-conducted unvoiced speech enhancement. IEEE Transactions on Audio Speech and Language Processing 20 9 (2012) 2505\u20132517.","DOI":"10.1109\/TASL.2012.2205241"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2009.4960405"},{"key":"e_1_3_3_2_51_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Jingxian Wang Chengfeng Pan Haojian Jin Vaibhav Singh Yash Jain Jason\u00a0I Hong Carmel Majidi and Swarun Kumar. 2019. RFID tattoo: A wireless platform for speech recognition. Proceedings of the ACM on Interactive Mobile Wearable and Ubiquitous Technologies 3 4 (2019) 1\u201324.","DOI":"10.1145\/3369812"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642092"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"crossref","unstructured":"Thomas Wolf Lysandre Debut Victor Sanh Julien Chaumond Clement Delangue Anthony Moi Pierric Cistac Tim Rault R\u00e9mi Louf Morgan Funtowicz et\u00a0al. 2019. Huggingface\u2019s transformers: State-of-the-art natural language processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.03771 (2019).","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676423"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1109\/WHC56415.2023.10224375"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"crossref","unstructured":"Shumin Zhai Michael Hunter and Barton\u00a0A Smith. 2002. Performance optimization of virtual keyboards. Human\u2013Computer Interaction 17 2-3 (2002) 229\u2013269.","DOI":"10.1207\/S15327051HCI172&3_4"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/642611.642630"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/3332165.3347924"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445166"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"crossref","unstructured":"Ruidong Zhang Mingyang Chen Benjamin Steeper Yaxuan Li Zihan Yan Yizhuo Chen Songyun Tao Tuochao Chen Hyunchul Lim and Cheng Zhang. 2021. SpeeChin: A smart necklace for silent speech recognition. Proceedings of the ACM on Interactive Mobile Wearable and Ubiquitous Technologies 5 4 (2021) 1\u201323.","DOI":"10.1145\/3494987"},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"publisher","DOI":"10.1145\/3490099.3511103"}],"event":{"name":"CUI '25: Proceedings of the 7th ACM Conference on Conversational User Interfaces","location":"Waterloo ON Canada","acronym":"CUI '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 7th ACM Conference on Conversational User Interfaces"],"original-title":[],"deposited":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T10:28:34Z","timestamp":1751711314000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3719160.3736612"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,7]]},"references-count":61,"alternative-id":["10.1145\/3719160.3736612","10.1145\/3719160"],"URL":"https:\/\/doi.org\/10.1145\/3719160.3736612","relation":{},"subject":[],"published":{"date-parts":[[2025,7,7]]},"assertion":[{"value":"2025-07-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}