{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:33:45Z","timestamp":1763202825742,"version":"3.28.0"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,4,14]]},"DOI":"10.1109\/icassp48485.2024.10448232","type":"proceedings-article","created":{"date-parts":[[2024,3,18]],"date-time":"2024-03-18T18:56:31Z","timestamp":1710788191000},"page":"10786-10790","source":"Crossref","is-referenced-by-count":3,"title":["GR0: Self-Supervised Global Representation Learning for Zero-Shot Voice Conversion"],"prefix":"10.1109","author":[{"given":"Yunyun","family":"Wang","sequence":"first","affiliation":[{"name":"Princeton University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaqi","family":"Su","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Adam","family":"Finkelstein","sequence":"additional","affiliation":[{"name":"Princeton University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeyu","family":"Jin","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"year":"2018","author":"Devlin","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","key":"ref1"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/TASLP.2021.3122291"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/taslp.2023.3288409"},{"key":"ref4","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Baevski"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.21437\/interspeech.2021-1280"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.21437\/Interspeech.2019-1873"},{"year":"2021","author":"Goyal","article-title":"Self-supervised pretraining of visual features in the wild","key":"ref7"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"ref9","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume-title":"ICML","author":"Chen"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/TKDE.2021.3090866"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1109\/ICASSP40776.2020.9054734"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/ICASSP43922.2022.9747590"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.21437\/Interspeech.2021-319"},{"key":"ref14","first-page":"2709","article-title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone","volume-title":"ICML","author":"Casanova"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.21437\/Interspeech.2020-1542"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.21437\/Interspeech.2021-475"},{"key":"ref17","first-page":"16251","article-title":"Neural analysis and synthesis: Reconstructing speech from self-supervised representations","volume-title":"Advances in Neural Information Processing Systems","volume":"34","author":"Choi"},{"key":"ref18","first-page":"5180","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","volume-title":"ICML","author":"Wang"},{"year":"2023","author":"Wang","article-title":"Neural codec language models are zero-shot text to speech synthesizers","key":"ref19"},{"key":"ref20","first-page":"17022","article-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Kong"},{"year":"2018","author":"Donahue","article-title":"Adversarial audio synthesis","key":"ref21"},{"year":"2017","author":"Lim","article-title":"Geometric gan","key":"ref22"},{"year":"2018","author":"Miyato","article-title":"Spectral normalization for generative adversarial networks","key":"ref23"},{"key":"ref24","first-page":"1558","article-title":"Autoencoding beyond pixels using a learned similarity metric","volume-title":"ICML","author":"Boesen"},{"key":"ref25","article-title":"Melgan: Generative adversarial networks for conditional waveform synthesis","volume-title":"Advances in neural information processing systems","volume":"32","author":"Kumar"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1109\/ICASSP.2018.8461329"},{"issue":"3","key":"ref27","first-page":"1638","article-title":"Swipe: a sawtooth waveform inspired pitch estimator for speech and music","volume":"124","author":"Harris","year":"2007","journal-title":"Journal of the Acoustical Society of America"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.18653\/v1\/2020.coling-main.519"},{"year":"2021","author":"Ravanelli","article-title":"SpeechBrain: A general-purpose speech toolkit","key":"ref29"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.21437\/Interspeech.2021-299"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.21437\/Interspeech.2022-405"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1109\/WASPAA52581.2021.9632770"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/ICASSP39728.2021.9413575"},{"doi-asserted-by":"publisher","key":"ref34","DOI":"10.1109\/LSP.2014.2379648"},{"year":"2019","author":"Yamagishi","journal-title":"Cstr vctk corpus: English multi-speaker corpus for cstr voice cloning toolkit (version 0.92)","key":"ref35"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.21437\/Interspeech.2020-2826"}],"event":{"name":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2024,4,14]]},"location":"Seoul, Korea, Republic of","end":{"date-parts":[[2024,4,19]]}},"container-title":["ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10445798\/10445803\/10448232.pdf?arnumber=10448232","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,3]],"date-time":"2024-08-03T04:46:04Z","timestamp":1722660364000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10448232\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,14]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/icassp48485.2024.10448232","relation":{},"subject":[],"published":{"date-parts":[[2024,4,14]]}}}