{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T10:25:20Z","timestamp":1758450320528,"version":"3.44.0"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783032049469"},{"type":"electronic","value":"9783032049476"}],"license":[{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-04947-6_59","type":"book-chapter","created":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T17:32:57Z","timestamp":1758389577000},"page":"619-629","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Speech Audio Generation from\u00a0Dynamic MRI via\u00a0a\u00a0Knowledge Enhanced Conditional Variational Autoencoder"],"prefix":"10.1007","author":[{"given":"Yaxuan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifei","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihua","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonghye","family":"Woo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fangxu","family":"Xing","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,21]]},"reference":[{"key":"59_CR1","doi-asserted-by":"crossref","unstructured":"Akbari, H., Arora, H., Cao, L., Mesgarani, N.: Lip2audspec: speech reconstruction from silent lip movements video. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2516\u20132520. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461856"},{"issue":"4","key":"59_CR2","doi-asserted-by":"publisher","first-page":"1791","DOI":"10.1121\/1.2335423","volume":"120","author":"E Bresch","year":"2006","unstructured":"Bresch, E., Nielsen, J., Nayak, K., Narayanan, S.: Synchronized and noise-robust audio recordings during realtime magnetic resonance imaging scans. J. Acoust. Soc. Am. 120(4), 1791\u20131794 (2006)","journal-title":"J. Acoust. Soc. Am."},{"key":"59_CR3","unstructured":"Caron, M., Misra, I., Mairal, J., Goyal, P., Bojanowski, P., Joulin, A.: Unsupervised learning of visual features by contrasting cluster assignments. Adv. Neural Inf. Process. Syst. 33, 9912\u20139924 (2020)"},{"key":"59_CR4","doi-asserted-by":"crossref","unstructured":"Chen, X., He, K.: Exploring simple Siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15750\u201315758 (2021)","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"59_CR5","doi-asserted-by":"crossref","unstructured":"Chi, T., Ru, P., Shamma, S.A.: Multiresolution spectrotemporal analysis of complex sounds. J. Acoust. Soc. Am. 118(2), 887\u2013906 (2005)","DOI":"10.1121\/1.1945807"},{"key":"59_CR6","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1016\/j.jspi.2017.01.004","volume":"185","author":"S Delattre","year":"2017","unstructured":"Delattre, S., Fournier, N.: On the kozachenko-leonenko entropy estimator. J. Stat. Plan. Inference 185, 69\u201393 (2017)","journal-title":"J. Stat. Plan. Inference"},{"issue":"6","key":"59_CR7","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.1109\/TBME.2013.2239293","volume":"60","author":"MA Ert\u00fcrk","year":"2013","unstructured":"Ert\u00fcrk, M.A., Bottomley, P.A., El-Sharkawy, A.M.M.: Denoising MRI using spectral subtraction. IEEE Trans. Biomed. Eng. 60(6), 1556\u20131562 (2013)","journal-title":"IEEE Trans. Biomed. Eng."},{"issue":"1","key":"59_CR8","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1186\/s41747-022-00293-x","volume":"6","author":"SM Jacobs","year":"2022","unstructured":"Jacobs, S.M., et al.: Image quality and subject experience of quiet t1-weighted 7-t brain imaging using a silent gradient coil. Eur. Radiol. Exp. 6(1), 36 (2022)","journal-title":"Eur. Radiol. Exp."},{"issue":"1","key":"59_CR9","doi-asserted-by":"publisher","first-page":"61","DOI":"10.1002\/mrm.29812","volume":"91","author":"R Jin","year":"2024","unstructured":"Jin, R., et al.: Optimization of 3D dynamic speech mri: poisson-disc undersampling and locally higher-rank reconstruction through partial separability model with regional optimized temporal basis. Magn. Reson. Med. 91(1), 61\u201374 (2024)","journal-title":"Magn. Reson. Med."},{"issue":"2","key":"59_CR10","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1002\/mrm.29486","volume":"89","author":"R Jin","year":"2023","unstructured":"Jin, R., et al.: Enhancing linguistic research through 2-mm isotropic 3d dynamic speech mri optimized by sparse temporal sampling and low-rank reconstruction. Magn. Reson. Med. 89(2), 652\u2013664 (2023)","journal-title":"Magn. Reson. Med."},{"key":"59_CR11","unstructured":"Kim, J., Kim, S., Kong, J., Yoon, S.: Glow-tts: a generative flow for text-to-speech via monotonic alignment search. Adv. Neural Inf. Process. Syst. 33, 8067\u20138077 (2020)"},{"key":"59_CR12","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In: International Conference on Machine Learning, pp. 5530\u20135540. PMLR (2021)"},{"key":"59_CR13","unstructured":"Kong, J., Kim, J., Bae, J.: Hifi-gan: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural Inf. Process. Syst. 33, 17022\u201317033 (2020)"},{"key":"59_CR14","unstructured":"Larsen, A.B.L., S\u00f8nderby, S.K., Larochelle, H., Winther, O.: Autoencoding beyond pixels using a learned similarity metric. In: International Conference on Machine Learning, pp. 1558\u20131566. PMLR (2016)"},{"key":"59_CR15","unstructured":"Lee, S.H., Kim, S.B., Lee, J.H., Song, E., Hwang, M.J., Lee, S.W.: Hierspeech: Bridging the gap between text and speech by hierarchical variational inference using self-supervised representations for speech synthesis. Adv. Neural Inf. Process. Syst. 35, 16624\u201316636 (2022)"},{"key":"59_CR16","doi-asserted-by":"crossref","unstructured":"Lim, Y., et al.: A multispeaker dataset of raw and reconstructed speech production real-time mri video and 3d volumetric images. Scientific Data 8(1), 187 (2021)","DOI":"10.1038\/s41597-021-00976-x"},{"issue":"3","key":"59_CR17","doi-asserted-by":"publisher","first-page":"1511","DOI":"10.1002\/mrm.27570","volume":"81","author":"Y Lim","year":"2019","unstructured":"Lim, Y., Zhu, Y., Lingala, S.G., Byrd, D., Narayanan, S., Nayak, K.S.: 3D dynamic mri of the vocal tract during natural speech. Magn. Reson. Med. 81(3), 1511\u20131520 (2019)","journal-title":"Magn. Reson. Med."},{"key":"59_CR18","doi-asserted-by":"publisher","unstructured":"Liu, X., et al.: Speech Audio Synthesis from Tagged MRI and Non-negative Matrix Factorization via Plastic Transformer. In: Greenspan, H., et al. (eds.) Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2023. MICCAI 2023. LNCS, vol. 14226, pp. 435\u2013445. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43990-2_41","DOI":"10.1007\/978-3-031-43990-2_41"},{"key":"59_CR19","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"59_CR20","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"59_CR21","doi-asserted-by":"publisher","first-page":"410","DOI":"10.1016\/j.wocn.2018.10.001","volume":"71","author":"SM Lulich","year":"2018","unstructured":"Lulich, S.M., Berkson, K.H., de Jong, K.: Acquiring and visualizing 3d\/4d ultrasound recordings of tongue motion. J. Phon. 71, 410\u2013424 (2018)","journal-title":"J. Phon."},{"key":"59_CR22","doi-asserted-by":"crossref","unstructured":"Mao, X., Li, Q., Xie, H., Lau, R.Y., Wang, Z., Paul\u00a0Smolley, S.: Least squares generative adversarial networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2794\u20132802 (2017)","DOI":"10.1109\/ICCV.2017.304"},{"key":"59_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13244-019-0754-2","volume":"10","author":"RM Mench\u00f3n-Lara","year":"2019","unstructured":"Mench\u00f3n-Lara, R.M., Simmross-Wattenberg, F., Casaseca-de-la Higuera, P., Mart\u00edn-Fern\u00e1ndez, M., Alberola-L\u00f3pez, C.: Reconstruction techniques for cardiac cine mri. Insights Imaging 10, 1\u201316 (2019)","journal-title":"Insights Imaging"},{"key":"59_CR24","unstructured":"Oquab, M., et\u00a0al.: Dinov2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"issue":"6","key":"59_CR25","doi-asserted-by":"publisher","first-page":"3078","DOI":"10.1121\/1.404204","volume":"92","author":"JS Perkell","year":"1992","unstructured":"Perkell, J.S., Cohen, M.H., Svirsky, M.A., Matthies, M.L., Garabieta, I., Jackson, M.T.: Electromagnetic midsagittal articulometer systems for transducing speech articulatory movements. J. Acoust. Soc. Am. 92(6), 3078\u20133096 (1992)","journal-title":"J. Acoust. Soc. Am."},{"key":"59_CR26","doi-asserted-by":"crossref","unstructured":"Prajwal, K., Mukhopadhyay, R., Namboodiri, V.P., Jawahar, C.: Learning individual speaking styles for accurate lip to speech synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13796\u201313805 (2020)","DOI":"10.1109\/CVPR42600.2020.01381"},{"key":"59_CR27","unstructured":"Recommendation, I.T.: Perceptual evaluation of speech quality (pesq): an objective method for end-to-end speech quality assessment of narrow-band telephone networks and speech codecs. Rec. ITU-T P. 862 (2001)"},{"key":"59_CR28","unstructured":"Rezende, D., Mohamed, S.: Variational inference with normalizing flows. In: International Conference on Machine Learning, pp. 1530\u20131538. PMLR (2015)"},{"key":"59_CR29","doi-asserted-by":"crossref","unstructured":"Toutios, A., et\u00a0al.: Illustrating the production of the international phonetic alphabet sounds using fast real-time magnetic resonance imaging. In: Interspeech, pp. 2428\u20132432 (2016)","DOI":"10.21437\/Interspeech.2016-605"},{"key":"59_CR30","doi-asserted-by":"crossref","unstructured":"Zheng, R.C., Ai, Y., Ling, Z.H.: Incorporating ultrasound tongue images for audiovisual speech enhancement. IEEE\/ACM Trans. Audio Speech Lang. Process. (2024)","DOI":"10.1109\/TASLP.2024.3361376"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-04947-6_59","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T17:33:08Z","timestamp":1758389588000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-04947-6_59"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,21]]},"ISBN":["9783032049469","9783032049476"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-04947-6_59","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025,9,21]]},"assertion":[{"value":"21 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Daejeon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.miccai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}