{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T17:10:04Z","timestamp":1756487404116,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T00:00:00Z","timestamp":1702684800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Humanities and Social Sciences Research Planning Fund of the Ministry of Education of China","award":["21A10011003"],"award-info":[{"award-number":["21A10011003"]}]},{"name":"Humanity and Social Science Youth Foundation of Ministry of Education of China","award":["21YJCZH117"],"award-info":[{"award-number":["21YJCZH117"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,16]]},"DOI":"10.1145\/3639592.3639623","type":"proceedings-article","created":{"date-parts":[[2024,4,13]],"date-time":"2024-04-13T12:04:31Z","timestamp":1713009871000},"page":"226-231","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-stage Multi-modalities Fusion of Lip, Tongue and Acoustics Information for Speech Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5266-6115","authenticated-orcid":false,"given":"Xuening","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer and Artificial Intelligence of Beijing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0743-0700","authenticated-orcid":false,"given":"Zhaopeng","family":"Qian","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence of Beijing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4234-1260","authenticated-orcid":false,"given":"Chongchong","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence of Beijing Technology and Business University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,4,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Gregory Hickok. 2012. Computational neuroanatomy of speech production. Nature reviews neuroscience 13 2 135-145.","DOI":"10.1038\/nrn3158"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6289039"},{"volume-title":"Speaker-independent silent speech recognition from flesh-point articulatory movements using an LSTM neural network","author":"Kim Myungjong","key":"e_1_3_2_1_3_1","unstructured":"Myungjong Kim, Beiming Cao, Ted Mau, and Jun Wang. 2017. Speaker-independent silent speech recognition from flesh-point articulatory movements using an LSTM neural network. IEEE\/ACM transactions on audio, speech, and language processing 25, 12, 2323-2336."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2011-111"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2003.1224072"},{"volume-title":"Seminars in speech and language","author":"Gibbon Fiona","key":"e_1_3_2_1_6_1","unstructured":"Fiona Gibbon and Alice Lee. 2015. Electropalatography for older children and adults with residual speech errors. In Seminars in speech and language, Vol. 36, 271-282."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053841"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6174"},{"key":"e_1_3_2_1_9_1","unstructured":"Xichen Pan Peiyu Chen Yichen Gong Helong Zhou Xinbing Wang and Zhouhan Lin. 2022. Leveraging unimodal self-supervised learning for multimodal audio-visual speech recognition. In arXiv preprint arXiv:2203.07996."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ejmp.2014.05.001"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2752365"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300376"},{"key":"e_1_3_2_1_13_1","volume-title":"G\u00e1bor Gosztolya.","author":"T\u00f3th L\u00e1szl\u00f3","year":"2023","unstructured":"L\u00e1szl\u00f3 T\u00f3th, Amin Honarmandi Shandiz, G\u00e1bor Gosztolya. 2023. Adaptation of tongue ultrasound-based silent speech interfaces using spatial transformer networks. In arXiv preprint arXiv:2305.19130."},{"key":"e_1_3_2_1_14_1","volume-title":"Lipnet: End-to-end sentence-level lipreading. In arXiv preprint arXiv:1611.01599.","author":"Assael Yannis M.","year":"2016","unstructured":"Yannis M. Assael, Brendan Shillingford, Shimon Whiteson, and Nando De Freitas. 2016. Lipnet: End-to-end sentence-level lipreading. In arXiv preprint arXiv:1611.01599."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/BIOCAS.2018.8584786"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2011-410"},{"key":"e_1_3_2_1_17_1","volume-title":"2018 19th International Carpathian Control Conference (ICCC), 98-103","author":"Czap L\u00e1szl\u00f3","year":"2018","unstructured":"Lu, Zhao, and L\u00e1szl\u00f3 Czap. 2018. Modelling the tongue movement of Chinese Shaanxi Xi'an dialect speech. In 2018 19th International Carpathian Control Conference (ICCC), 98-103."},{"issue":"10","key":"e_1_3_2_1_18_1","first-page":"48","article-title":"Quick review of human speech production mechanism","volume":"9","author":"Mahendru Harish Chander","year":"2014","unstructured":"Harish Chander Mahendru. 2014. Quick review of human speech production mechanism. International Journal of Engineering Research and Development, 9(10), 48-54.","journal-title":"International Journal of Engineering Research and Development"},{"key":"e_1_3_2_1_19_1","unstructured":"B. Denby J. Cai P. Roussel L. Crevier-Buchman S. Manitsaris G. Chollet M. Stone and C. Pillot. 2013. The silent speech challenge archive. https:\/\/ftp.espci.fr\/pub\/sigma\/"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.08.002"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference","author":"Kimura Naoki","year":"2022","unstructured":"Naoki Kimura, Zixiong Su, Takaaki Saeki, and Jun Rekimoto. 2022. SSR7000: A synchronized corpus of ultrasound tongue imaging for end-to-end silent speech recognition. In Proceedings of the Thirteenth Language Resources and Evaluation Conference, Marseille, France: European Language Resources Association, 6866\u20136873."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383619"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Manuel Sam Ribeiro Aciel Eshky Korin Richmond and Steve Renals. 2021. Silent versus modal multi-speaker speech recognition from ultrasound and video. In arXiv preprint arXiv:2103.00333.","DOI":"10.21437\/Interspeech.2021-23"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.11.004"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2018.02.002"},{"volume-title":"Transfer learning from adult to children for speech recognition: Evaluation, analysis and recommendations. Computer speech & language","author":"Shivakumar Prashanth Gurunath","key":"e_1_3_2_1_26_1","unstructured":"Prashanth Gurunath Shivakumar, and Panayiotis Georgiou. 2020. Transfer learning from adult to children for speech recognition: Evaluation, analysis and recommendations. Computer speech & language, Vol.63, 101077."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2015.7415532"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2022.103148"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1006"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414460"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Jonathan Boigne Biman Liyanage and Ted \u00d6strem. 2020. Recognizing more emotions with less data using self-supervised transfer learning.\u201d In arXiv preprint arXiv:2011.05585.","DOI":"10.20944\/preprints202008.0645.v1"},{"key":"e_1_3_2_1_32_1","first-page":"3370","article-title":"Temporal context in speech emotion recognition In Proc","volume":"2021","author":"Xia Yangyang","year":"2021","unstructured":"Yangyang Xia, Li-Wei Chen, Alexander Rudnicky, and Richard M. Stern. 2021. Temporal context in speech emotion recognition In Proc. Interspeech 2021, 3370\u20133374.","journal-title":"Interspeech"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095036"},{"key":"e_1_3_2_1_34_1","first-page":"1509","article-title":"Exploring wav2vec 2.0 on speaker verification and language identification","volume":"2021","author":"Fan Zhiyun","year":"2021","unstructured":"Zhiyun Fan, Meng Li, Shiyu Zhou, and Bo Xu. 2021. Exploring wav2vec 2.0 on speaker verification and language identification. In Proc. Interspeech 2021, 1509\u20131513.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_35_1","unstructured":"Bowen Shi Wei-Ning Hsu Kushal Lakhotia and Abdelrahman Mohamed. 2022. Learning audio-visual speech representation by masked multimodal cluster prediction. In arXiv preprint arXiv:2201.02184."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSRE.2023.3262001"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.241"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2019.07.011"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Taku Kudo. 2018. Subword regularization: Improving neural network translation models with multiple subword candidates. In arXiv preprint arXiv:1804.10959.","DOI":"10.18653\/v1\/P18-1007"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2013-48"},{"key":"e_1_3_2_1_41_1","volume-title":"IEEE 2011 workshop on automatic speech recognition and understanding, no. CONF. IEEE Signal Processing Society.","author":"Povey Daniel","year":"2011","unstructured":"Daniel Povey, Arnab Ghoshal, Gilles Boulianne, Lukas Burget, Ondrej Glembek, Nagendra Goel, Mirko Hannemann, Petr Motlicek, Yanmin Qian, Petr Schwarz, Jan Silovsky, Georg Stemmer, Karel Vesely. 2011.The kaldi speech recognition toolkit. In IEEE 2011 workshop on automatic speech recognition and understanding, no. CONF. IEEE Signal Processing Society."}],"event":{"name":"AICCC 2023: 2023 6th Artificial Intelligence and Cloud Computing Conference","acronym":"AICCC 2023","location":"Kyoto Japan"},"container-title":["2023 6th Artificial Intelligence and Cloud Computing Conference (AICCC)"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3639592.3639623","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3639592.3639623","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T16:41:42Z","timestamp":1756485702000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3639592.3639623"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,16]]},"references-count":41,"alternative-id":["10.1145\/3639592.3639623","10.1145\/3639592"],"URL":"https:\/\/doi.org\/10.1145\/3639592.3639623","relation":{},"subject":[],"published":{"date-parts":[[2023,12,16]]},"assertion":[{"value":"2024-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}