{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T09:30:42Z","timestamp":1787650242284,"version":"build-2736575974"},"reference-count":53,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Institute of Information and communications Technology Planning and Evaluation"},{"name":"Korea government","award":["2021-0-00456"],"award-info":[{"award-number":["2021-0-00456"]}]},{"name":"Development of Ultra-high Speech Quality Technology for Remote Multi-speaker Conference System"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/taslp.2024.3364085","type":"journal-article","created":{"date-parts":[[2024,2,8]],"date-time":"2024-02-08T18:55:02Z","timestamp":1707418502000},"page":"1519-1530","source":"Crossref","is-referenced-by-count":6,"title":["Transfer Learning for Low-Resource, Multi-Lingual, and Zero-Shot Multi-Speaker Text-to-Speech"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4689-3110","authenticated-orcid":false,"given":"Myeonghun","family":"Jeong","sequence":"first","affiliation":[{"name":"Kakao Enterprise, Seongnam, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8150-765X","authenticated-orcid":false,"given":"Minchan","family":"Kim","sequence":"additional","affiliation":[{"name":"Institute of New Media and Communications, Department of Electrical and Computer Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1319-8215","authenticated-orcid":false,"given":"Byoung Jin","family":"Choi","sequence":"additional","affiliation":[{"name":"Institute of New Media and Communications, Department of Electrical and Computer Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9978-0582","authenticated-orcid":false,"given":"Jaesam","family":"Yoon","sequence":"additional","affiliation":[{"name":"Kakao Enterprise, Seongnam, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4711-780X","authenticated-orcid":false,"given":"Won","family":"Jang","sequence":"additional","affiliation":[{"name":"Kakao Enterprise, Seongnam, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0568-4902","authenticated-orcid":false,"given":"Nam Soo","family":"Kim","sequence":"additional","affiliation":[{"name":"Institute of New Media and Communications, Department of Electrical and Computer Engineering, Seoul National University, Seoul, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-469"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref3","article-title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ren","year":"2021"},{"key":"ref4","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim","year":"2021"},{"key":"ref5","first-page":"8067","article-title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Kim","year":"2020"},{"key":"ref6","first-page":"741","article-title":"Low-resource multilingual and zero-shot multispeaker TTS","volume-title":"Proc. 2nd Conf. Asia-Pacific Chapter Assoc. Comput. Linguistics 12th Int. Joint Conf. Natural Lang. Process.","author":"Lux","year":"2022"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3141200"},{"key":"ref8","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Casanova","year":"2022"},{"key":"ref9","first-page":"5410","article-title":"Almost unsupervised text to speech and automatic speech recognition","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ren","year":"2019"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403331"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-11071"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-816"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-225"},{"key":"ref14","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Baevski","year":"2020"},{"key":"ref15","first-page":"27826","article-title":"Unsupervised speech recognition","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Baevski","year":"2021"},{"key":"ref16","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Jia","year":"2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054535"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1774"},{"key":"ref19","first-page":"7748","article-title":"Meta-stylespeech: Multi-speaker adaptive text-to-speech generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Min","year":"2021"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-441"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-901"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2022.3226655"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.23919\/APSIPAASC55919.2022.9979900"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1229"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682927"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2679"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-2668"},{"key":"ref29","article-title":"Cross-lingual multi-speaker text-to-speech synthesis for voice cloning without using parallel corpus for unseen speakers","author":"Liu","year":"2019"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1632"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-46"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682674"},{"key":"ref33","article-title":"Multilingual byte2speech models for scalable low-resource speech synthesis","author":"He","year":"2021"},{"key":"ref34","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-475"},{"key":"ref38","article-title":"Neural analysis and synthesis: Reconstructing speech from self-supervised representations","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Choi","year":"2021"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10797"},{"key":"ref40","first-page":"16624","article-title":"HierSpeech: Bridging the gap between text and speech by hierarchical variational inference using self-supervised representations for speech synthesis","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Lee","year":"2022"},{"key":"ref41","first-page":"18003","article-title":"ContentVec: An improved self-supervised speech representation by disentangling speakers","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2022"},{"key":"ref42","article-title":"NANSY++: Unified voice synthesis with neural analysis and synthesis","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Choi","year":"2023"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1016\/j.wocn.2018.07.001"},{"key":"ref44","volume-title":"The Art of VA Filter Design","author":"Zavalishin","year":"2012"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-329"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1016\/S1364-6613(99)01294-2"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2826"},{"key":"ref49","article-title":"CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92)","author":"Yamagishi","year":"2019"},{"key":"ref50","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2023"},{"key":"ref51","article-title":"Clova baseline system for the voxceleb speaker recognition challenge 2020","author":"Heo","year":"2020"},{"key":"ref52","article-title":"Speechbrain: A general-purpose speech toolkit","author":"Ravanelli","year":"2021"},{"key":"ref53","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/10304349\/10428082.pdf?arnumber=10428082","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,14]],"date-time":"2024-03-14T05:21:57Z","timestamp":1710393717000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10428082\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":53,"URL":"https:\/\/doi.org\/10.1109\/taslp.2024.3364085","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}