{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:24:22Z","timestamp":1782314662774,"version":"3.54.5"},"reference-count":118,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"name":"Amity University Dubai, United Arab Emirates"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/access.2025.3605236","type":"journal-article","created":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T17:31:52Z","timestamp":1756834312000},"page":"155729-155758","source":"Crossref","is-referenced-by-count":11,"title":["Advancing Text-to-Speech Systems for Low-Resource Languages: Challenges, Innovations, and Future Directions"],"prefix":"10.1109","volume":"13","author":[{"given":"Shashi","family":"Bhushan","sequence":"first","affiliation":[{"name":"Computer Science Department, Amity University Dubai, Dubai, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ved","family":"Prakash Mishra","sequence":"additional","affiliation":[{"name":"Computer Science Department, Amity University Dubai, Dubai, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2451-4949","authenticated-orcid":false,"given":"Vinay","family":"Rishiwal","sequence":"additional","affiliation":[{"name":"Department of CSIT, MJP Rohilkhand University, Bareilly, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9871-3871","authenticated-orcid":false,"given":"Sharmila","family":"Arunkumar","sequence":"additional","affiliation":[{"name":"Department of EC, Raj Kumar Goel Institute of Technology, Ghaziabad, Uttar Pradesh, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4353-0274","authenticated-orcid":false,"given":"Udit","family":"Agarwal","sequence":"additional","affiliation":[{"name":"Department of CSIT, RBMI Group of Institutions, Bareilly, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2025.3527745"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref4","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","author":"Ren","year":"2020","journal-title":"arXiv:2006.04558"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.5120\/ijca2015906965"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/BRACIS.2019.00025"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.5120\/ijca2016907992"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.12720\/jait.13.5.398-412"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-1288"},{"key":"ref10","first-page":"164","article-title":"Wolfgang von kempelen\u2019s speaking machine as an instrument for demonstration and research","volume-title":"Proc. 17th Int. Congr. Phonetic Sci.","author":"Trouvain"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2025.3535414"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref13","article-title":"Deep text-to-speech system with Seq2Seq model","author":"Wang","year":"2019","journal-title":"arXiv:1903.07398"},{"key":"ref14","first-page":"7586","article-title":"Non-autoregressive neural text-to-speech","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Peng"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Blizzard.2021-4"},{"key":"ref16","article-title":"VARA-TTS: Non-autoregressive text-to-speech synthesis based on very deep VAE with residual attention","author":"Liu","year":"2021","journal-title":"arXiv:2102.06431"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-471"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-52"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383629"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413969"},{"key":"ref21","first-page":"3918","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Oord"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3440637"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639187"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3499741"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638996"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2016.2516032"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-134"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-246"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854321"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.1992.226124"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/MELCON.2000.879998"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/WCCIT.2013.6618665"},{"key":"ref36","first-page":"261","article-title":"Combining a vector space representation of linguistic context with a deep neural network for text-to-speech synthesis","volume-title":"Proc. 8th ISCA Speech Synth. Workshop","author":"Lu"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178813"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953089"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCSPA49915.2021.9385731"},{"key":"ref40","article-title":"Finnish end-to-end speech synthesis with tacotron 2 and wavenet","author":"Alastalo","year":"2021"},{"issue":"4","key":"ref41","first-page":"111","article-title":"An end-to-end Chinese speech synthesis scheme based on tacotron 2","volume":"2019","author":"Wang","year":"2019","journal-title":"J. East China Normal Univ. (Natural Sci.)"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICACI52617.2021.9435882"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2017.8282234"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2018.8706713"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TELFOR48224.2019.8971351"},{"key":"ref46","article-title":"Deep voice 2: Multi-speaker neural text-to-speech","author":"Arik","year":"2017","journal-title":"arXiv:1705.08947"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683682"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-24"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2020.09.003"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1598"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2872060"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-443"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462473"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2761547"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953090"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-444"},{"key":"ref57","article-title":"AdaSpeech: Adaptive text to speech for custom voice","author":"Chen","year":"2021","journal-title":"arXiv:2103.00993"},{"key":"ref58","first-page":"7748","article-title":"Meta-stylespeech: Multi-speaker adaptive text-to-speech generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Min"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3167258"},{"key":"ref60","first-page":"1","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ye"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2096"},{"key":"ref62","article-title":"Neural codec language models are zero-shot text to speech synthesizers","author":"Wang","year":"2023","journal-title":"arXiv:2301.02111"},{"key":"ref63","article-title":"PromptTTS 2: Describing and generating voices with text prompt","author":"Leng","year":"2023","journal-title":"arXiv:2309.02285"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00618"},{"key":"ref65","article-title":"ClariNet: Parallel wave generation in end-to-end text-to-speech","author":"Ping","year":"2018","journal-title":"arXiv:1807.07281"},{"key":"ref66","article-title":"Char2wav: End-to-end speech synthesis","author":"Sotelo"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682770"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3139"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2679"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-2024"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472738"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICTC49870.2020.9289277"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58309-5_22"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11719-w"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746883"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1251"},{"key":"ref77","first-page":"4693","article-title":"Towards end-to-end prosody transfer for expressive speech synthesis with tacotron","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Skerry-Ryan"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-10761"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095131"},{"key":"ref80","article-title":"RNN approaches to text normalization: A challenge","author":"Sproat","year":"2016","journal-title":"arXiv:1611.00068"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1061"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-1028"},{"key":"ref83","first-page":"2410","article-title":"Efficient neural audio synthesis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kalchbrenner"},{"key":"ref84","article-title":"FloWaveNet: A generative flow for raw audio","author":"Kim","year":"2018","journal-title":"arXiv:1811.02155"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414444"},{"key":"ref86","article-title":"Universal MelGAN: A robust neural vocoder for high-fidelity waveform generation in multiple domains","author":"Jang","year":"2020","journal-title":"arXiv:2011.09631"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3288409"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404780"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682368"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462678"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1235"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1094"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1208"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053512"},{"key":"ref96","article-title":"Fast wavenet generation algorithm","author":"Le Paine","year":"2016","journal-title":"arXiv:1611.09482"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2786"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383549"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-41"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413466"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10338"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746291"},{"key":"ref106","article-title":"Text-to-speech data augmentation for low resource speech recognition","author":"Zevallos","year":"2022","journal-title":"arXiv:2204.00291"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO58844.2023.10289912"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-021-00225-4"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1229"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3364085"},{"key":"ref111","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Casanova"},{"key":"ref112","first-page":"16624","article-title":"Hierspeech: Bridging the gap between text and speech by hierarchical variational inference using self-supervised representations for speech synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Lee"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403331"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2664"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-29516-5_5"},{"key":"ref116","article-title":"End-to-end text-to-speech for low-resource languages by cross-lingual transfer learning","author":"Tu","year":"2019","journal-title":"arXiv:1904.06508"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682168"},{"key":"ref118","first-page":"2005","article-title":"TTS for low resource languages: A Bangla synthesizer","volume-title":"Proc. 10th Int. Conf. Lang. Resour. Eval. (LREC)","author":"Gutkin"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10820123\/11146779.pdf?arnumber=11146779","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T18:44:28Z","timestamp":1763750668000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11146779\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":118,"URL":"https:\/\/doi.org\/10.1109\/access.2025.3605236","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}