{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T06:13:49Z","timestamp":1771049629965,"version":"3.50.1"},"reference-count":45,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Guangdong Provincial Key Laboratory of Artificial Intelligence in Medical Image Analysis and Application","award":["2022B1212010011"],"award-info":[{"award-number":["2022B1212010011"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput. Soc. Syst."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1109\/tcss.2025.3603008","type":"journal-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:42:18Z","timestamp":1760722938000},"page":"856-867","source":"Crossref","is-referenced-by-count":0,"title":["SSL-VC: One-Shot Voice Conversion Through Self-Supervised Learning"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3935-9482","authenticated-orcid":false,"given":"Chenglong","family":"Jiang","sequence":"first","affiliation":[{"name":"GHT Company Ltd., Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0726-012X","authenticated-orcid":false,"given":"Linrong","family":"Pan","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8925-8192","authenticated-orcid":false,"given":"Ying","family":"Gao","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8212-9751","authenticated-orcid":false,"given":"Kuanghua","family":"Su","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5953-9163","authenticated-orcid":false,"given":"Gaoze","family":"Hou","sequence":"additional","affiliation":[{"name":"School of Environmental and Ecological Engineering, Dalian University of Technology, Dalian, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4952-699X","authenticated-orcid":false,"given":"Xiping","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Medical Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2663"},{"key":"ref2","first-page":"5210","article-title":"AutoVC: Zero-shot voice style transfer with only autoencoder loss","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2019"},{"key":"ref3","first-page":"7836","article-title":"Unsupervised speech decomposition via triple information bottleneck","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2020"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096220"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3257839"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10797"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-489"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref10","article-title":"DinoSR: Self-distillation and online clustering for self-supervised speech representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Liu","year":"2024"},{"key":"ref11","article-title":"A survey on neural speech synthesis","author":"Tan","year":"2021"},{"key":"ref12","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim","year":"2021"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-838"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095776"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-571"},{"key":"ref16","first-page":"16251","article-title":"Neural analysis and synthesis: Reconstructing speech from self-supervised representations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Choi","year":"2021"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP57327.2022.10037890"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2022.3203888"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1608"},{"key":"ref20","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref22","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Baevski","year":"2020"},{"key":"ref23","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Baevski","year":"2020"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383605"},{"key":"ref25","first-page":"18003","article-title":"Contentvec: An improved self-supervised speech representation by disentangling speakers","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2022"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2016.7820786"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1830"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2917232"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-99-8145-8_37"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613800"},{"key":"ref31","first-page":"294","article-title":"Voicemixer: Adversarial voice style mixup","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Lee","year":"2021"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3076867"},{"key":"ref33","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Casanova","year":"2022"},{"key":"ref34","article-title":"Diffusion-based voice conversion with fast maximum likelihood sampling scheme","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Popov","year":"2022"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445804"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-475"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095191"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096742"},{"key":"ref40","article-title":"AdaSpeech: Adaptive text to speech for custom voice","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Chen","year":"2021"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-283"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-299"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-00296-0_5"},{"key":"ref45","article-title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Ren","year":"2021"}],"container-title":["IEEE Transactions on Computational Social Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6570650\/11395558\/11206317.pdf?arnumber=11206317","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T05:42:30Z","timestamp":1771047750000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11206317\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":45,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tcss.2025.3603008","relation":{},"ISSN":["2329-924X","2373-7476"],"issn-type":[{"value":"2329-924X","type":"electronic"},{"value":"2373-7476","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2]]}}}