{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T05:24:59Z","timestamp":1755926699082,"version":"3.28.0"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,7,18]]},"DOI":"10.1109\/ijcnn55064.2022.9892405","type":"proceedings-article","created":{"date-parts":[[2022,9,30]],"date-time":"2022-09-30T19:56:04Z","timestamp":1664567764000},"page":"01-08","source":"Crossref","is-referenced-by-count":6,"title":["Zero-shot Voice Conversion via Self-supervised Prosody Representation Learning"],"prefix":"10.1109","author":[{"given":"Shijun","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science University of St. Gallen,AIML Lab,St. Gallen,Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Damian","family":"Borth","sequence":"additional","affiliation":[{"name":"School of Computer Science University of St. Gallen,AIML Lab,St. Gallen,Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"van der maaten","year":"2008","journal-title":"Journal of Machine Learning Research"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"year":"2017","author":"veaux","journal-title":"CSTR VCTK corpus English multi-speaker corpus for cstr voice cloning toolkit","key":"ref31"},{"key":"ref30","volume":"absi1807 3748","author":"van den oord","year":"2018","journal-title":"Representation learning with contrastive predictive coding"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/ICASSP39728.2021.9414257"},{"key":"ref11","first-page":"6284","article-title":"FO-consistent many-to-many non-parallel voice conversion via conditional autoencoder","author":"kaizhi","year":"0","journal-title":"ICASSP 2020 &#x2013; 2020 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/TASLP.2020.3038524"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1109\/ICASSP.2018.8461375"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1109\/ICASSP40776.2020.9053854"},{"key":"ref15","volume":"abs 2006 4154","author":"wu","year":"2020","journal-title":"Vqvc+ One-shot voice conversion by vector quantization and u-net architecture"},{"key":"ref16","article-title":"Neural discrete representation learning","author":"van den oord","year":"2017","journal-title":"NIPS"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1109\/ICASSP39728.2021.9414975"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.21437\/Interspeech.2020-2861"},{"key":"ref19","first-page":"6189","article-title":"Mel-lotron: Multispeaker expressive voice synthesis by conditioning on rhythm, pitch and global style tokens","author":"rafael","year":"0","journal-title":"ICASSP 2020 &#x2013; 2020 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref28","article-title":"Deep relative attributes","author":"souri","year":"2016","journal-title":"ACCV"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1109\/ISCSLP.2018.8706604"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1145\/3490354.3494373"},{"key":"ref3","volume":"abs 1711 11293","author":"kaneko","year":"2017","journal-title":"Parallel-data-free voice conversion using cycle-consistent adversarial networks"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/SLT.2018.8639535"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.21437\/Interspeech.2020-1693"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"ref8","volume":"abs 1905 5879","author":"kaizhi","year":"2019","journal-title":"Zero-shot voice style transfer with only autoencoder loss"},{"key":"ref7","article-title":"Blow: a single-scale hyperconditioned flow for non-parallel raw-audio voice conversion","author":"serra","year":"2019","journal-title":"NeurIPS"},{"key":"ref2","first-page":"1403","article-title":"Voice conversion based on speaker-dependent restricted boltzmann machines","volume":"97 d","author":"toru","year":"2014","journal-title":"IEICE Trans Inf Syst"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.21437\/Interspeech.2019-2663"},{"key":"ref1","first-page":"4859","article-title":"Modulation spectrum-constrained trajectory training algorithm for gmm-based voice conversion","author":"shinnosuke","year":"0","journal-title":"2015 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref20","volume":"abs 2004 11284","author":"qian","year":"2020","journal-title":"Unsupervised speech decomposition via triple information bottleneck"},{"key":"ref22","volume":"abs 2002 5709","author":"chen","year":"2020","journal-title":"A simple framework for contrastive learning of visual representations"},{"key":"ref21","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","author":"wang","year":"2018","journal-title":"ICML"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.21437\/Interspeech.2019-1873"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref26","volume":"abs 2104 6074","author":"wang","year":"2021","journal-title":"Noisevc Towards high quality zero-shot voice conversion"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/SLT48900.2021.9383605"}],"event":{"name":"2022 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2022,7,18]]},"location":"Padua, Italy","end":{"date-parts":[[2022,7,23]]}},"container-title":["2022 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9891857\/9889787\/09892405.pdf?arnumber=9892405","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,4]],"date-time":"2022-11-04T01:26:53Z","timestamp":1667525213000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9892405\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,18]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/ijcnn55064.2022.9892405","relation":{},"subject":[],"published":{"date-parts":[[2022,7,18]]}}}