{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T23:19:01Z","timestamp":1784675941798,"version":"3.55.0"},"reference-count":55,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"name":"Faculty of Computer Science, Universitas Indonesia","award":["NKB-008\/UN2.F11.D\/HKP.05.00\/2023"],"award-info":[{"award-number":["NKB-008\/UN2.F11.D\/HKP.05.00\/2023"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/access.2024.3396377","type":"journal-article","created":{"date-parts":[[2024,5,2]],"date-time":"2024-05-02T17:28:23Z","timestamp":1714670903000},"page":"63528-63547","source":"Crossref","is-referenced-by-count":13,"title":["Zero-Shot Voice Cloning Text-to-Speech for Dysphonia Disorder Speakers"],"prefix":"10.1109","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3217-7025","authenticated-orcid":false,"given":"Kurniawati","family":"Azizah","sequence":"first","affiliation":[{"name":"Faculty of Computer Science, Universitas Indonesia, Depok, Indonesia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.11.001"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3093"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/mc.2019.2934592"},{"key":"ref4","volume-title":"Speech and Language Processing: An Introduction To Natural Language Processing, Computational Linguistics, and Speech Recognition","author":"Jurasfky","year":"2009"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462529"},{"key":"ref7","first-page":"6270","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","author":"Van Den Oord"},{"key":"ref8","first-page":"264","article-title":"Deep voice: Real-time neural text-to-speech","volume-title":"Proc. 34th Int. Conf. Mach. Learn. (ICML)","author":"Arik"},{"key":"ref9","first-page":"2963","article-title":"Deep voice 2: Multi-speaker neural text-to-speech","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst. (NIPS)","author":"Gibiansky"},{"key":"ref10","first-page":"1","article-title":"Deep voice 3: Scaling text-to-speech with convolutional sequence learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Ping"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682353"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/taslp.2019.2935807"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3390\/info10040131"},{"key":"ref16","first-page":"1","article-title":"FastSpeech2: Fast and high-quality end-to-end text to speech","volume-title":"Proc. 9th Int. Conf. Learn. Represent. (ICLR)","author":"Ren"},{"key":"ref17","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume":"139","author":"Kim","year":"2021","journal-title":"Proc. PMLR"},{"key":"ref18","first-page":"5932","article-title":"Fitting new speakers based on a short untranscribed sample","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","author":"Nachmani"},{"key":"ref19","article-title":"Multi-speaker end-to-end speech synthesis","author":"Park","year":"2019","journal-title":"arXiv:1907.04462"},{"key":"ref20","first-page":"1","article-title":"Sample efficient adaptive text-to-speech","volume-title":"Proc. 7th Int. Conf. Learn. Represent.","author":"Chen"},{"key":"ref21","article-title":"Modeling multi-speaker latent space to improve neural TTS: Quick enrolling new speaker and enhancing premium voice","author":"Deng","year":"2018","journal-title":"arXiv:1812.05253"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1558"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639659"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/ssw.2019-5"},{"key":"ref25","first-page":"1","article-title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/icassp39728.2021.9413889"},{"key":"ref27","first-page":"4485","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","volume-title":"Proc. 32nd Conf. Neural Inf. Process. Syst. (NIPS)","author":"Jia"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054535"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-1032"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/access.2022.3141200"},{"key":"ref31","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","author":"Casanova","year":"2022","journal-title":"Proc. ICML"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.2015.7178964"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/icassp40776.2020.9054596"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.cmpb.2021.106602"},{"key":"ref36","article-title":"Cross-lingual multi-speaker text-to-speech synthesis for voice cloning without using parallel corpus for unseen speakers","author":"Liu","year":"2019","journal-title":"arXiv:1911.11601"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21428\/594757db.1bcc4f0c"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2205"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-1891"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC47483.2019.9023039"},{"key":"ref42","article-title":"Clova baseline system for the VoxCeleb speaker recognition challenge 2020","author":"Soo Heo","year":"2020","journal-title":"arXiv:2009.14153"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref45","volume-title":"Uncommon Voice","author":"Moore","year":"2024"},{"key":"ref46","volume-title":"Tacotron 2 Without WaveNet","year":"2020"},{"key":"ref47","volume-title":"Speaker Encoder","year":"2020"},{"key":"ref48","volume-title":"WaveGlow","year":"2020"},{"key":"ref49","first-page":"1","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. 3rd Int. Conf. Learn. Represent.","author":"Kingma"},{"key":"ref50","volume-title":"Fundamentals of Speech Recognition","author":"Rabiner","year":"1993"},{"key":"ref51","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume":"202","author":"Radford","year":"2023","journal-title":"Proc. Mach. Learn. Res."},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2003.1318504"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-2003"},{"key":"ref54","article-title":"UMAP: Uniform manifold approximation and projection for dimension reduction","author":"McInnes","year":"2018","journal-title":"arXiv:1802.03426"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/access.2023.3276480"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10380310\/10517609.pdf?arnumber=10517609","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T05:23:25Z","timestamp":1715232205000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10517609\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":55,"URL":"https:\/\/doi.org\/10.1109\/access.2024.3396377","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}