{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T19:44:26Z","timestamp":1779306266341,"version":"3.51.4"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T00:00:00Z","timestamp":1764720000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T00:00:00Z","timestamp":1767744000000},"content-version":"vor","delay-in-days":35,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"Information & Communications Technology Planning & Evaluation","award":["RS-2024-00438027"],"award-info":[{"award-number":["RS-2024-00438027"]}]},{"name":"Information & Communications Technology Planning & Evaluation","award":["RS-2024-00438027"],"award-info":[{"award-number":["RS-2024-00438027"]}]},{"name":"Information & Communications Technology Planning & Evaluation","award":["RS-2025-25441313"],"award-info":[{"award-number":["RS-2025-25441313"]}]},{"name":"Information & Communications Technology Planning & Evaluation","award":["RS-2025-25441313"],"award-info":[{"award-number":["RS-2025-25441313"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"DOI":"10.1186\/s13636-025-00438-x","type":"journal-article","created":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T07:15:00Z","timestamp":1764746100000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multilingual speech-to-vocal tract visualization using deep learning for pronunciation training"],"prefix":"10.1186","volume":"2026","author":[{"given":"Rodrigo","family":"Picinini M\u00e9xas","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunji","family":"Chu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5539-9520","authenticated-orcid":false,"given":"Unsang","family":"Park","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,3]]},"reference":[{"key":"438_CR1","doi-asserted-by":"crossref","unstructured":"X.\u00a0Wu, P.\u00a0Hu, Y.\u00a0Wu, X.\u00a0Lyu, Y.P. Cao, Y.\u00a0Shan, W.\u00a0Yang, Z.\u00a0Sun, X.\u00a0Qi, in Proceedings of the IEEE\/CVF International Conference on Computer Vision, Speech2lip: High-fidelity speech to lip generation by learning from a short video (2023), pp. 22168\u201322177","DOI":"10.1109\/ICCV51070.2023.02026"},{"key":"438_CR2","doi-asserted-by":"crossref","unstructured":"K.\u00a0Prajwal, R.\u00a0Mukhopadhyay, V.P. Namboodiri, C.\u00a0Jawahar, in Proceedings of the 28th ACM international conference on multimedia, A lip sync expert is all you need for speech to lip generation in the wild (2020), pp. 484\u2013492","DOI":"10.1145\/3394171.3413532"},{"key":"438_CR3","doi-asserted-by":"crossref","unstructured":"K.C. Wang, J.\u00a0Zhang, J.\u00a0Huang, Q.\u00a0Li, M.T. Sun, K.\u00a0Sakai, W.S. Ku, in 2023 IEEE International Conference on Smart Computing (SMARTCOMP), Ca-wav2lip: Coordinate attention-based speech to lip synthesis in the wild (IEEE, 2023), pp. 1\u20138","DOI":"10.1109\/SMARTCOMP58114.2023.00018"},{"key":"438_CR4","first-page":"2758","volume":"34","author":"M Kim","year":"2021","unstructured":"M. Kim, J. Hong, Y.M. Ro, Lip to speech synthesis with visual context attentional GAN. Adv. Neural. Inf. Process. Syst. 34, 2758\u20132770 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"438_CR5","doi-asserted-by":"crossref","unstructured":"A.\u00a0Richard, M.\u00a0Zollh\u00f6fer, Y.\u00a0Wen, F.\u00a0De\u00a0la Torre, Y.\u00a0Sheikh, in Proceedings of the IEEE\/CVF International Conference on Computer Vision, Meshtalk: 3d face animation from speech using cross-modality disentanglement (2021), pp. 1173\u20131182","DOI":"10.1109\/ICCV48922.2021.00121"},{"key":"438_CR6","doi-asserted-by":"crossref","unstructured":"K.I. Haque, Z.\u00a0Yumak, in Proceedings of the 25th International Conference on Multimodal Interaction, Facexhubert: Text-less speech-driven e (x) pressive 3d facial animation synthesis using self-supervised speech representation learning (2023), pp. 282\u2013291","DOI":"10.1145\/3577190.3614157"},{"key":"438_CR7","doi-asserted-by":"crossref","unstructured":"J.\u00a0Xing, M.\u00a0Xia, Y.\u00a0Zhang, X.\u00a0Cun, J.\u00a0Wang, T.T. Wong, in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Codetalker: Speech-driven 3d facial animation with discrete motion prior (2023), pp. 12,780\u201312,790","DOI":"10.1109\/CVPR52729.2023.01229"},{"key":"438_CR8","doi-asserted-by":"crossref","unstructured":"J.\u00a0Wang, X.\u00a0Qian, M.\u00a0Zhang, R.T. Tan, H.\u00a0Li, in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seeing what you said: Talking face generation guided by a lip reading expert (2023), pp. 14,653\u201314,662","DOI":"10.1109\/CVPR52729.2023.01408"},{"key":"438_CR9","doi-asserted-by":"crossref","unstructured":"X.\u00a0Zhang, J.\u00a0Wang, N.\u00a0Cheng, E.\u00a0Xiao, J.\u00a0Xiao, in Asia-pacific web (APWeb) and web-age information management (WAIM) joint international conference on web and big data, Shallow diffusion motion model for talking face generation from speech (Springer, 2022), pp. 144\u2013157","DOI":"10.1007\/978-3-031-25198-6_11"},{"issue":"1","key":"438_CR10","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-025-00403-8","volume":"2025","author":"C Yu","year":"2025","unstructured":"C. Yu, X. Wang, Z. Qian, Silent speech recognition using visual cascading fusion of tongue-lip movements based on pre-trained and fine-tuned model. EURASIP J. Audio Speech Music Process. 2025(1), 16 (2025). https:\/\/doi.org\/10.1186\/s13636-025-00403-8","journal-title":"EURASIP J. Audio Speech Music Process."},{"issue":"1","key":"438_CR11","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-024-00345-7","volume":"2024","author":"D Gimeno-G\u00f3mez","year":"2024","unstructured":"D. Gimeno-G\u00f3mez, C.D. Mart\u00ednez-Hinarejos, Continuous lipreading based on acoustic temporal alignments. EURASIP J. Audio Speech Music Process. 2024(1), 25 (2024). https:\/\/doi.org\/10.1186\/s13636-024-00345-7","journal-title":"EURASIP J. Audio Speech Music Process."},{"key":"438_CR12","doi-asserted-by":"publisher","unstructured":"K.\u00a0Xu, M.\u00a0Feng, W.\u00a0Huang, in Proceedings of the 30th ACM International Conference on Multimedia, Seeing speech: Magnetic resonance imaging-based vocal tract deformation visualization using cross-modal transformer (Association for Computing Machinery, New York, NY, USA, 2022), MM \u201922, pp. 6947\u20136949. https:\/\/doi.org\/10.1145\/3503161.3547728","DOI":"10.1145\/3503161.3547728"},{"key":"438_CR13","doi-asserted-by":"publisher","DOI":"10.3390\/jimaging9100233","author":"K Isaieva","year":"2023","unstructured":"K. Isaieva, F. Odille, Y. Laprie, G. Drouot, J. Felblinger, P. Vuissoz, Super-resolved dynamic 3d reconstruction of the vocal tract during natural speech. J. Imaging (2023). https:\/\/doi.org\/10.3390\/jimaging9100233","journal-title":"J. Imaging"},{"key":"438_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.specom.2022.04.004","volume":"141","author":"V Ribeiro","year":"2022","unstructured":"V. Ribeiro, K. Isaieva, J. Leclere, P. Vuissoz, Y. Laprie, Automatic generation of the complete vocal tract shape from the sequence of phonemes to be articulated. Speech Communication 141, 1\u201313 (2022). https:\/\/doi.org\/10.1016\/j.specom.2022.04.004","journal-title":"Speech Communication"},{"issue":"5","key":"438_CR15","doi-asserted-by":"publisher","first-page":"838","DOI":"10.1109\/TMI.2012.2230017","volume":"32","author":"Y Zhu","year":"2013","unstructured":"Y. Zhu, Y. Kim, M.I. Proctor, S.S. Narayanan, K.S. Nayak, Dynamic 3-d visualization of vocal tract shaping during speech. IEEE Trans. Med. Imaging 32(5), 838\u2013848 (2013). https:\/\/doi.org\/10.1109\/TMI.2012.2230017","journal-title":"IEEE Trans. Med. Imaging"},{"key":"438_CR16","volume-title":"in 22nd International Congress on Acoustics (ICA), Copy synthesis of running speech based on vocal tract imaging and audio recording","author":"B Elie","year":"2016","unstructured":"B. Elie, Y. Laprie, in 22nd International Congress on Acoustics (ICA), Copy synthesis of running speech based on vocal tract imaging and audio recording (Buenos Aires, Argentina, 2016)"},{"key":"438_CR17","doi-asserted-by":"publisher","unstructured":"P.\u00a0Wu, L.\u00a0Chen, C.\u00a0Cho, S.\u00a0Watanabe, L.\u00a0Goldstein, A.W. Black, G.K. Anumanchipalli, in ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Speaker-independent acoustic-to-articulatory speech inversion (2023), pp. 1\u20135. https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10096796","DOI":"10.1109\/ICASSP49357.2023.10096796"},{"key":"438_CR18","doi-asserted-by":"publisher","unstructured":"Y.\u00a0Xu, English speech recognition and evaluation of pronunciation quality using deep learning. Mobile Information Systems 2022(1), 7186,375 (2022). https:\/\/doi.org\/10.1155\/2022\/7186375","DOI":"10.1155\/2022\/7186375"},{"key":"438_CR19","unstructured":"D.\u00a0Korzekwa, B.\u00a0Kostek, R.\u00a0Barra-Chicote, Automated detection of pronunciation errors in non-native english speech employing deep learning. Doctoral Dissertation (2022). https:\/\/www.amazon.science\/publications\/automated-detection-of-pronunciation-errors-in-non-native-english-speech-employing-deep-learning"},{"key":"438_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2023.103009","volume":"156","author":"C Zhu","year":"2024","unstructured":"C. Zhu, A. Wumaier, D. Wei, Z. Fan, J. Yang, H. Yu, Z. Kadeer, L. Wang, Pronunciation error detection model based on feature fusion. Speech Communication 156, 103,009 (2024). https:\/\/doi.org\/10.1016\/j.specom.2023.103009","journal-title":"Speech Communication"},{"key":"438_CR21","doi-asserted-by":"publisher","unstructured":"M.H. Mozaffari, W.S. Lee, in 2021 IEEE International Conference on Bioinformatics and Biomedicine (BIBM), Second language pronunciation training by ultrasound-enhanced visual augmented reality (2021), pp. 3043\u20133050. https:\/\/doi.org\/10.1109\/BIBM52615.2021.9669622","DOI":"10.1109\/BIBM52615.2021.9669622"},{"issue":"1","key":"438_CR22","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1038\/s41597-021-00976-x","volume":"8","author":"Y Lim","year":"2021","unstructured":"Y. Lim, A. Toutios, Y. Bliesener, Y. Tian, S.G. Lingala, C. Vaz, T. Sorensen, M. Oh, S. Harper, W. Chen, Y. Lee, J. T\u00f6ger, M.L. Monteserin, C. Smith, B. Godinez, L. Goldstein, D. Byrd, K.S. Nayak, S.S. Narayanan, A multispeaker dataset of raw and reconstructed speech production real-time MRI video and 3d volumetric images. Sci. Data 8(1), 187 (2021). https:\/\/doi.org\/10.1038\/s41597-021-00976-x","journal-title":"Sci. Data"},{"issue":"3","key":"438_CR23","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1121\/1.4890284","volume":"136","author":"S Narayanan","year":"2014","unstructured":"S. Narayanan, A. Toutios, V. Ramanarayanan, A. Lammert, J. Kim, S. Lee, K. Nayak, Y.C. Kim, Y. Zhu, L. Goldstein, D. Byrd, E. Bresch, P. Ghosh, A. Katsamanis, M. Proctor, Real-time magnetic resonance imaging and electromagnetic articulography database for speech production research (TC). J. Acoust. Soc. Am. 136(3), 1307 (2014). https:\/\/doi.org\/10.1121\/1.4890284","journal-title":"J. Acoust. Soc. Am."},{"key":"438_CR24","doi-asserted-by":"publisher","unstructured":"I.K. Douros, J.\u00a0Felblinger, J.\u00a0Frahm, K.\u00a0Isaieva, A.A. Joseph, Y.\u00a0Laprie, F.\u00a0Odille, A.\u00a0Tsukanova, D.\u00a0Voit, P.A. Vuissoz, in Interspeech 2019, A multimodal real-time mri articulatory corpus of french for speech research (2019), pp. 1556\u20131560. https:\/\/doi.org\/10.21437\/Interspeech.2019-1700","DOI":"10.21437\/Interspeech.2019-1700"},{"key":"438_CR25","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0060603","author":"P Birkholz","year":"2013","unstructured":"P. Birkholz, Modeling consonant-vowel coarticulation for articulatory speech synthesis. PLoS One (2013). https:\/\/doi.org\/10.1371\/journal.pone.0060603","journal-title":"PLoS One"},{"key":"438_CR26","first-page":"12,449","volume":"33","author":"A Baevski","year":"2020","unstructured":"A. Baevski, Y. Zhou, A. Mohamed, M. Auli, Wav2vec 2.0: a framework for self-supervised learning of speech representations. Adv. Neural Inf. Process. Syst. 33, 12,449-12,460 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"438_CR27","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W Hsu","year":"2021","unstructured":"W. Hsu, B. Bolte, Y.H. Tsai, K. Lakhotia, R. Salakhutdinov, A. Mohamed, Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3451\u20133460 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"438_CR28","unstructured":"J.\u00a0Chung, C.\u00a0Gulcehre, K.\u00a0Cho, Y.\u00a0Bengio, in NIPS 2014 Workshop on Deep Learning, December 2014, Empirical evaluation of gated recurrent neural networks on sequence modeling (2014)"},{"key":"438_CR29","unstructured":"R.P. M\u00e9xas. Multilingual Speech-To-Vocal Tract Visualization (2024). https:\/\/watching-sounds.github.io\/MultiSpeechToVocalTract\/. Accessed 1 Dec 2024"},{"key":"438_CR30","unstructured":"K.\u00a0Hyang-hee, Voice onset time(vot) during korean plosives production : A preliminary study on normal and apraxia of speech subjects. J Korean Soc Laryngol Phoniatr Logop 8(1), 49\u201353 (1997). http:\/\/jkslp.org\/journal\/view.php?number=1626. http:\/\/jkslp.org\/journal\/view.php?number=1626"},{"key":"438_CR31","unstructured":"T.B. Brown, Language models are few-shot learners. arXiv preprint arXiv:2005.14165 (2020)"},{"key":"438_CR32","unstructured":"Bootphon. Phonemizer (2024). https:\/\/github.com\/bootphon\/phonemizer. Accessed 3 Aug 2024"},{"key":"438_CR33","doi-asserted-by":"publisher","unstructured":"V.\u00a0Panayotov, G.\u00a0Chen, D.\u00a0Povey, S.\u00a0Khudanpur, in 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Librispeech: An asr corpus based on public domain audio books (2015), pp. 5206\u20135210. https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"438_CR34","doi-asserted-by":"crossref","unstructured":"H.\u00a0Choi, S.\u00a0Lee, S.\u00a0Lee, Diff-hiervc: Diffusion-based hierarchical voice conversion with robust pitch generation and masked prior for zero-shot speaker adaptation. International Speech Communication Association pp. 2283\u20132287 (2023)","DOI":"10.21437\/Interspeech.2023-817"},{"key":"438_CR35","doi-asserted-by":"publisher","unstructured":"S.\u00a0Medina, D.\u00a0Tome, C.\u00a0Stoll, M.\u00a0Tiede, K.\u00a0Munhall, A.\u00a0Hauptmann, I.\u00a0Matthews, in 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), Speech driven tongue animation (2022), pp. 20,374\u201320,384. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01976","DOI":"10.1109\/CVPR52688.2022.01976"},{"key":"438_CR36","doi-asserted-by":"publisher","unstructured":"C.\u00a0Cho, P.\u00a0Wu, A.\u00a0Mohamed, G.K. Anumanchipalli, in ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Evidence of vocal tract articulation in self-supervised learning of speech (IEEE, 2023). https:\/\/doi.org\/10.1109\/icassp49357.2023.10094711","DOI":"10.1109\/icassp49357.2023.10094711"},{"key":"438_CR37","unstructured":"P.\u00a0D\u2019hers. gTTS: Google Text-to-Speech Python Library (2023). Version 2.5.3. https:\/\/github.com\/pndurette\/gTTS. Accessed 24 Aug 2024"},{"key":"438_CR38","unstructured":"R.\u00a0Ardila, M.\u00a0Branson, K.\u00a0Davis, M.\u00a0Henretty, M.\u00a0Kohler, J.\u00a0Meyer, R.\u00a0Morais, L.\u00a0Saunders, F.M. Tyers, G.\u00a0Weber, Common voice: A massively-multilingual speech corpus. arXiv preprint arXiv:1912.06670 (2019)"},{"key":"438_CR39","unstructured":"R.P. M\u00e9xas. Software of Multilingual Speech-To-Vocal Tract Visualization (2024). https:\/\/zenodo.org\/records\/15510606. Accessed 25 May 2025"},{"key":"438_CR40","unstructured":"R.P. M\u00e9xas. Dataset of Multilingual Speech-To-Vocal Tract Visualization (2024). https:\/\/zenodo.org\/records\/15510640. Accessed 25 May 2025"}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00438-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-025-00438-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00438-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T13:22:58Z","timestamp":1767792178000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1186\/s13636-025-00438-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,3]]},"references-count":40,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["438"],"URL":"https:\/\/doi.org\/10.1186\/s13636-025-00438-x","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,3]]},"assertion":[{"value":"3 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This survey study involved voluntary participation from language learners, with no collection of personal or identifiable information and no medical or clinical intervention. All participants were informed of the purpose of the study and provided consent prior to participation. According to institutional and national research ethics guidelines, studies involving anonymous survey data with minimal risk do not require formal ethics committee approval. Therefore, no Institutional Review Board approval was sought for this research.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare that they have no competing interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"3"}}