{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:03:15Z","timestamp":1750309395549,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Nature Science Foundation of China","award":["U23B2053"],"award-info":[{"award-number":["U23B2053"]}]},{"name":"National Nature Science Foundation of China","award":["62301521"],"award-info":[{"award-number":["62301521"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100020593","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["WK2100000033"],"award-info":[{"award-number":["WK2100000033"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100020593","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Anhui Provincial Natural Science Foundation","award":["2308085QF200"],"award-info":[{"award-number":["2308085QF200"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680770","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"6559-6568","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Speech Reconstruction from Silent Lip and Tongue Articulation by Diffusion Models and Text-Guided Pseudo Target Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8074-9553","authenticated-orcid":false,"given":"Rui-Chen","family":"Zheng","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6668-022X","authenticated-orcid":false,"given":"Yang","family":"Ai","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7853-5273","authenticated-orcid":false,"given":"Zhen-Hua","family":"Ling","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proc. ICLR","author":"Chen Nanxin","year":"2020","unstructured":"Nanxin Chen, Yu Zhang, Heiga Zen, Ron J Weiss, Mohammad Norouzi, and William Chan. 2020. WaveGrad: Estimating Gradients for Waveform Generation. In Proc. ICLR 2020."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-54427-4_19"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1031"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-939"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.08.002"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2738564"},{"key":"e_1_3_2_1_7_1","volume-title":"Jos\u00e9 L P\u00e9rez-C\u00f3rdoba, and Angel M Gomez.","author":"Gonzalez-Lopez Jose A","year":"2020","unstructured":"Jose A Gonzalez-Lopez, Alejandro Gomez-Alanis, Juan M Mart\u00edn Do\u00f1as, Jos\u00e9 L P\u00e9rez-C\u00f3rdoba, and Angel M Gomez. 2020. Silent speech interfaces for speech restoration: A review. IEEE access 8 (2020), 177995--178021."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461732"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19966"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611787"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548081"},{"key":"e_1_3_2_1_12_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems 33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020), 6840--6851."},{"key":"e_1_3_2_1_13_1","volume-title":"Neural dubber: Dubbing for videos according to scripts. Advances in neural information processing systems 34","author":"Hu Chenxu","year":"2021","unstructured":"Chenxu Hu, Qiao Tian, Tingle Li, Wang Yuping, Yuxuan Wang, and Hang Zhao. 2021. Neural dubber: Dubbing for videos according to scripts. Advances in neural information processing systems 34 (2021), 16582--16595."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547855"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2015.03.005"},{"key":"e_1_3_2_1_16_1","volume-title":"Proc. Interspeech","author":"Hueber Thomas","year":"2011","unstructured":"Thomas Hueber, Elie-Laurent Benaroya, Bruce Denby, and G\u00e9rard Chollet. 2011. Statistical mapping between articulatory and acoustic data for an ultrasoundbased silent speech interface. In Proc. Interspeech 2011. 593--596."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-469"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.1715112"},{"key":"e_1_3_2_1_19_1","first-page":"2758","article-title":"Lip to speech synthesis with visual context attentional gan","volume":"34","author":"Kim Minsu","year":"2021","unstructured":"Minsu Kim, Joanna Hong, and Yong Man Ro. 2021. Lip to speech synthesis with visual context attentional gan. In Advances in Neural Information Processing Systems, Vol. 34. 2758--2770.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095582"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300376"},{"key":"e_1_3_2_1_22_1","volume-title":"Proc. ICLR","author":"Kong Zhifeng","year":"2020","unstructured":"Zhifeng Kong, Wei Ping, Jiaji Huang, Kexin Zhao, and Bryan Catanzaro. 2020. DiffWave: A Versatile Diffusion Model for Audio Synthesis. In Proc. ICLR 2020."},{"key":"e_1_3_2_1_23_1","volume-title":"Deep speaker: an end-to-end neural speaker embedding system. arXiv preprint arXiv:1705.02304","author":"Li Chao","year":"2017","unstructured":"Chao Li, Xiaokong Ma, Bing Jiang, Xiangang Li, Xuewei Zhang, Xiao Liu, Ying Cao, Ajay Kannan, and Zhenyao Zhu. 2017. Deep speaker: an end-to-end neural speaker embedding system. arXiv preprint arXiv:1705.02304 (2017)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21350"},{"key":"e_1_3_2_1_25_1","volume-title":"Diffgan-tts: High-fidelity and efficient text-to-speech with denoising diffusion gans. arXiv preprint arXiv:2201.11972","author":"Liu Songxiang","year":"2022","unstructured":"Songxiang Liu, Dan Su, and Dong Yu. 2022. Diffgan-tts: High-fidelity and efficient text-to-speech with denoising diffusion gans. arXiv preprint arXiv:2201.11972 (2022)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-715"},{"key":"e_1_3_2_1_27_1","first-page":"1541","article-title":"MOSNet","volume":"2019","author":"Lo Chen-Chou","year":"2019","unstructured":"Chen-Chou Lo, Szu-Wei Fu, Wen-Chin Huang, Xin Wang, Junichi Yamagishi, Yu Tsao, and Hsin-Min Wang. 2019. MOSNet: Deep Learning-Based Objective Assessment for Voice Conversion. In Proc. Interspeech 2019. 1541--1545.","journal-title":"Deep Learning-Based Objective Assessment for Voice Conversion. In Proc. Interspeech"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746421"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2179"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746901"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054057"},{"key":"e_1_3_2_1_33_1","volume-title":"Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499","author":"van den Oord Aaron","year":"2016","unstructured":"Aaron van den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, and Koray Kavukcuoglu. 2016. Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 (2016)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413532"},{"key":"e_1_3_2_1_35_1","volume-title":"Proc. ICLR","author":"Ren Yi","year":"2020","unstructured":"Yi Ren, Chenxu Hu, Xu Tan, Tao Qin, Sheng Zhao, Zhou Zhao, and Tie-Yan Liu. 2020. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. In Proc. ICLR 2020."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-23"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383619"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2752365"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_1_40_1","volume-title":"Proc. ICLR","author":"Song Yang","year":"2020","unstructured":"Yang Song, Jascha Sohl-Dickstein, Diederik P Kingma, Abhishek Kumar, Stefano Ermon, and Ben Poole. 2020. Score-Based Generative Modeling through Stochastic Differential Equations. In Proc. ICLR 2020."},{"key":"e_1_3_2_1_41_1","volume-title":"Proc. International Congress of Phonetic Sciences","author":"Teplansky Kristin J","year":"2019","unstructured":"Kristin J Teplansky, Brian Y Tsang, and Jun Wang. 2019. Tongue and lip motion patterns in voiced, whispered, and silent vowel production. In Proc. International Congress of Phonetic Sciences 2019. 1--5."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2854"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1078"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548194"},{"key":"e_1_3_2_1_45_1","first-page":"2207","article-title":"ESPNet","volume":"2018","author":"Watanabe Shinji","year":"2018","unstructured":"Shinji Watanabe, Takaaki Hori, Shigeki Karita, Tomoki Hayashi, Jiro Nishitoba, Yuya Unno, Nelson Enrique Yalta Soplin, Jahn Heymann, Matthew Wiesner, Nanxin Chen, et al. 2018. ESPNet: End-to-End Speech Processing Toolkit. In Proc. Interspeech 2018. 2207-2211.","journal-title":"End-to-End Speech Processing Toolkit. In Proc. Interspeech"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096064"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17693"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096920"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680770","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680770","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:42Z","timestamp":1750294662000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680770"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":50,"alternative-id":["10.1145\/3664647.3680770","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680770","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}