{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,2]],"date-time":"2025-12-02T15:06:26Z","timestamp":1764687986117,"version":"3.37.3"},"reference-count":72,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100000275","name":"Leverhulme Trust Research Project","doi-asserted-by":"publisher","award":["RPG-2019-241"],"award-info":[{"award-number":["RPG-2019-241"]}],"id":[{"id":"10.13039\/501100000275","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/access.2022.3166922","type":"journal-article","created":{"date-parts":[[2022,4,12]],"date-time":"2022-04-12T19:33:04Z","timestamp":1649791984000},"page":"41489-41502","source":"Crossref","is-referenced-by-count":4,"title":["Estimating Underlying Articulatory Targets of Thai Vowels by Using Deep Learning Based on Generating Synthetic Samples From a 3D Vocal Tract Model and Data Augmentation"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9820-1980","authenticated-orcid":false,"given":"Thanat","family":"Lapthawan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0869-089X","authenticated-orcid":false,"given":"Santitham","family":"Prom-On","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0167-8123","authenticated-orcid":false,"given":"Peter","family":"Birkholz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8541-2658","authenticated-orcid":false,"given":"Yi","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref72","article-title":"Using self-supervised learning can improve model robustness and uncertainty","author":"hendrycks","year":"2019","journal-title":"arXiv 1906 12340"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref70","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1660160"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ECTICon.2015.7206999"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref32","first-page":"21","article-title":"Vocal tract length perturbation (VTLP) improves speech recognition","volume":"117","author":"jaitly","year":"2013","journal-title":"Proc ICML Workshop Deep Learn Audio Speech Lang"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054130"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-711"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003990"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1993.319366"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1986.1168657"},{"key":"ref60","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref61","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014","journal-title":"arXiv 1412 6980"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1007\/s00365-006-0663-2"},{"key":"ref27","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","article-title":"Deep learning","volume":"521","author":"lecun","year":"2015","journal-title":"Nature"},{"key":"ref64","first-page":"341","article-title":"Praat, a system for doing phonetics by computer","volume":"5","author":"boersma","year":"2001","journal-title":"Glot Int"},{"key":"ref65","article-title":"UMAP: Uniform manifold approximation and projection for dimension reduction","author":"mcinnes","year":"2018","journal-title":"arXiv 1802 03426"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2017.08.003"},{"journal-title":"Handbook of the International Phonetic Association A Guide to the Use of the International Phonetic Alphabet","year":"1999","key":"ref66"},{"journal-title":"Acoustic Theory of Speech Production","year":"1970","author":"fant","key":"ref67"},{"key":"ref68","first-page":"16","article-title":"The tones of central thai: Some perceptual experiments","volume":"1","author":"abramson","year":"1975","journal-title":"Studies in Thai linguistics in honor of William J Gedney"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1017\/S0025100300004746"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2019.02777"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1121\/1.1913427"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178812"},{"key":"ref22","first-page":"4854","article-title":"Articulatory movement prediction using deep bidirectional long short-term memory based recurrent neural networks and word\/phone embeddings","author":"li","year":"2015","journal-title":"Proc 16th Annu Conf Int Speech Commun Assoc"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-493"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178893"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682506"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2017.8282219"},{"key":"ref25","first-page":"1","article-title":"A deep neural network for acoustic-articulatory speech inversion","author":"uria","year":"2011","journal-title":"Proc NIPS Workshop Deep Learn Unsupervised Feature Learn"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.889731"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1121\/1.3037222"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00033-8"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1980.1163420"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"ref54","first-page":"377","article-title":"Simulation of vocal tract growth for articulatory speech synthesis","author":"birkholz","year":"2007","journal-title":"Proc 16th Int Congr Phonetic Sci"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2091632"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2008-342"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1186\/1687-4722-2014-23"},{"key":"ref11","first-page":"746","article-title":"A rough guide to the acoustic-to-articulatory inversion of speech","author":"toutios","year":"2003","journal-title":"Proc 6th Hellenic Eur Conf Comput Math Appl"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1334"},{"key":"ref12","first-page":"74","article-title":"An empirical investigation of the non-uniqueness in the acoustic-to-articulatory mapping","author":"qin","year":"2007","journal-title":"Proc 8th Annu Conf Int Speech Commun Assoc"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.07.005"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(97)00021-6"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1177\/002383099303600303"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2019.05.004"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1121\/1.1921448"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2003.822636"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2004-410"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-260"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1093\/jmt\/47.1.2"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2014796"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2005.1415149"},{"key":"ref8","first-page":"1","article-title":"Applying articulatory features to telephone-based speaker verification","volume":"1","author":"leung","year":"2004","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref49","first-page":"493","article-title":"Vocal tract model adaptation using magnetic resonance imaging","author":"birkholz","year":"2006","journal-title":"Proc 7th Int Seminar Speech Prod (ISSP)"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639267"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2010-196"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21248\/zaspil.40.2005.258"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/DEVLRN.2014.6982981"},{"journal-title":"VocalTractLab-Towards High-quality Articulatory Speech Synthesis","year":"2017","author":"birkholz","key":"ref48"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0060603"},{"key":"ref42","first-page":"159","article-title":"Training a vocal tract synthesiser to imitate speech using distal supervised learning","volume":"2","author":"howard","year":"2005","journal-title":"Proc 10th Int Conf Speech Comput"},{"key":"ref41","first-page":"52","article-title":"Articulatory copy synthesis using long-short term memory networks","author":"yingming","year":"2020","journal-title":"Studientexte Zur Sprachkommunikation Elektronische Sprachsignalverarbeitung"},{"key":"ref44","first-page":"304","article-title":"Modelling vowel acquisition using the birkholz synthesizer","author":"howard","year":"2019","journal-title":"Sprachkommunikation Elektronische Sprachsignalverarbeitung"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1121\/1.3514544"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9668973\/09755932.pdf?arnumber=9755932","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,26]],"date-time":"2022-04-26T16:07:06Z","timestamp":1650989226000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9755932\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":72,"URL":"https:\/\/doi.org\/10.1109\/access.2022.3166922","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2022]]}}}