{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,11]],"date-time":"2025-04-11T11:29:43Z","timestamp":1744370983166,"version":"3.37.3"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,4,14]]},"DOI":"10.1109\/icassp48485.2024.10447224","type":"proceedings-article","created":{"date-parts":[[2024,3,18]],"date-time":"2024-03-18T18:56:31Z","timestamp":1710788191000},"page":"2170-2174","source":"Crossref","is-referenced-by-count":1,"title":["An Audio-Textual Diffusion Model for Converting Speech Signals into Ultrasound Tongue Imaging Data"],"prefix":"10.1109","author":[{"given":"Yudong","family":"Yang","sequence":"first","affiliation":[{"name":"Shenzhen Institute of Advanced Technology,Chinese Academy of Sciences,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rongfeng","family":"Su","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology,Chinese Academy of Sciences,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaokang","family":"Liu","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology,Chinese Academy of Sciences,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nan","family":"Yan","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology,Chinese Academy of Sciences,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lan","family":"Wang","sequence":"additional","affiliation":[{"name":"Shenzhen Institute of Advanced Technology,Chinese Academy of Sciences,Shenzhen,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2752365"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3133218"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.ultrasmedbio.2021.06.009"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2021.02.001"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383619"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096920"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094797"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2019.8851769"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471970"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.3791\/55123"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094703"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1381"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-174"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43999-5_14"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3390\/e25101469"},{"key":"ref16","first-page":"36479","article-title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","volume-title":"NeurIPS 2022","volume":"35","author":"Saharia"},{"key":"ref17","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"NeurIPS 2020","volume":"33","author":"Baevski"},{"article-title":"Bert: Pretraining of deep bidirectional transformers for language understanding","year":"2018","author":"Devlin","key":"ref18"},{"article-title":"Video diffusion models","year":"2022","author":"Ho","key":"ref19"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"ref21","article-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium","volume-title":"NeurIPS 2017","volume":"30","author":"Heusel"},{"key":"ref22","first-page":"26565","article-title":"Elucidating the design space of diffusion-based generative models","volume-title":"NeurIPS 2022","volume":"35","author":"Karras"},{"article-title":"Score-based generative modeling through stochastic differential equations","volume-title":"ICLR 2021","author":"Song","key":"ref23"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2014","author":"Simonyan","key":"ref25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00361"}],"event":{"name":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2024,4,14]]},"location":"Seoul, Korea, Republic of","end":{"date-parts":[[2024,4,19]]}},"container-title":["ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10445798\/10445803\/10447224.pdf?arnumber=10447224","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,2]],"date-time":"2024-08-02T05:32:44Z","timestamp":1722576764000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10447224\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,14]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/icassp48485.2024.10447224","relation":{},"subject":[],"published":{"date-parts":[[2024,4,14]]}}}