{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T12:23:36Z","timestamp":1742991816059,"version":"3.40.3"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030661502"},{"type":"electronic","value":"9783030661519"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-66151-9_5","type":"book-chapter","created":{"date-parts":[[2020,12,21]],"date-time":"2020-12-21T00:02:47Z","timestamp":1608508967000},"page":"69-84","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["StarGAN-ZSVC: Towards Zero-Shot Voice Conversion in Low-Resource Contexts"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3001-6292","authenticated-orcid":false,"given":"Matthew","family":"Baas","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2980-3475","authenticated-orcid":false,"given":"Herman","family":"Kamper","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,12,21]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Cho, K., et al.: Learning phrase representations using RNN encoder-decoder for statistical machine translation. In: EMNLP (2014)","DOI":"10.3115\/v1\/D14-1179"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Choi, Y., Choi, M., Kim, M., Ha, J.W., Kim, S., Choo, J.: StarGAN: unified generative adversarial networks for multi-domain image-to-image translation. In: IEEE CVPR (2018)","DOI":"10.1109\/CVPR.2018.00916"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Chorowski, J., Weiss, R.J., Bengio, S., van den Oord, A.: Unsupervised speech representation learning using WaveNet autoencoders. arXiv e-prints arXiv:1901.08810 (2019)","DOI":"10.1109\/TASLP.2019.2938863"},{"key":"5_CR4","unstructured":"Dauphin, Y.N., Fan, A., Auli, M., Grangier, D.: Language modeling with gated convolutional networks. In: Precup, D., Teh, Y.W. (eds.) PMLR (2017)"},{"key":"5_CR5","unstructured":"Dumoulin, V., Shlens, J., Kudlur, M.: A learned representation for artistic style. In: ICLR (2017)"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Erro, D., Moreno, A.: Weighted frequency warping for voice conversion. In: INTERSPEECH (2007)","DOI":"10.21437\/Interspeech.2007-550"},{"key":"5_CR7","doi-asserted-by":"crossref","unstructured":"He, T., Zhang, Z., Zhang, H., Zhang, Z., Xie, J., Li, M.: Bag of tricks for image classification with convolutional neural networks. In: IEEE CVPR (2019)","DOI":"10.1109\/CVPR.2019.00065"},{"key":"5_CR8","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9, 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"5_CR9","unstructured":"Howard, J., Gugger, S.: DynamicUnet: create a U-Net from a given architecture (2020). https:\/\/docs.fast.ai\/vision.models.unet#DynamicUnet. Accessed 8 Aug 2020"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Huang, C., Lin, Y.Y., Lee, H., Lee, L.: Defending Your Voice: Adversarial Attack on Voice Conversion. arXiv e-prints arXiv:2005.08781 (2020)","DOI":"10.1109\/SLT48900.2021.9383529"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Kameoka, H., Kaneko, T., Tanaka, K., Hojo, N.: StarGAN-VC: non-parallel many-to-many voice conversion using star generative adversarial networks. In: IEEE SLT Workshop (2018)","DOI":"10.1109\/SLT.2018.8639535"},{"issue":"9","key":"5_CR12","doi-asserted-by":"publisher","first-page":"1432","DOI":"10.1109\/TASLP.2019.2917232","volume":"27","author":"H Kameoka","year":"2019","unstructured":"Kameoka, H., Kaneko, T., Tanaka, K., Hojo, N.: ACVAE-VC: non-parallel voice conversion with auxiliary classifier variational autoencoder. IEEE Trans. Audio Speech Lang. Process. 27(9), 1432\u20131443 (2019)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Kaneko, T., Kameoka, H., Tanaka, K., Hojo, N.: StarGAN-VC2: rethinking conditional methods for StarGAN-based voice conversion. In: INTERSPEECH (2019)","DOI":"10.21437\/Interspeech.2019-2236"},{"key":"5_CR14","unstructured":"Kingma, D.P., Ba, J.: Adam: A Method for Stochastic Optimization. arXiv e-prints arXiv:1412.6980 (2014)"},{"key":"5_CR15","unstructured":"Kumar, K., et al.: MelGAN: generative adversarial networks for conditional waveform synthesis. In: NeurIPS (2019)"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Lal Srivastava, B.M., Vauquier, N., Sahidullah, M., Bellet, A., Tommasi, M., Vincent, E.: Evaluating voice conversion-based privacy protection against informed attackers. In: ICASSP (2020)","DOI":"10.1109\/ICASSP40776.2020.9053868"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Lorenzo-Trueba, J., et al.: The voice conversion challenge 2018: promoting development of parallel and nonparallel methods. In: Odyssey Speaker and Language Recognition Workshop (2018)","DOI":"10.21437\/Odyssey.2018-28"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Mao, X., Li, Q., Xie, H., Lau, R.Y., Wang, Z., Smolley, S.P.: Least squares generative adversarial networks. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.304"},{"key":"5_CR19","unstructured":"Miyato, T., Koyama, M.: cGANs with projection discriminator. In: ICLR (2018)"},{"issue":"7","key":"5_CR20","doi-asserted-by":"publisher","first-page":"1877","DOI":"10.1587\/transinf.2015EDP7457","volume":"E99.D","author":"M Morise","year":"2016","unstructured":"Morise, M., Yokomori, F., Ozawa, K.: WORLD: a vocoder-based high-quality speech synthesis system for real-time applications. IEICE Trans. Inf. Syst. E99.D(7), 1877\u20131884 (2016)","journal-title":"IEICE Trans. Inf. Syst."},{"key":"5_CR21","unstructured":"van den Oord, A., et al.: WaveNet: A Generative Model for Raw Audio. arXiv e-prints arXiv:1609.03499 (2016)"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: an ASR corpus based on public domain audio books. In: IEEE ICASSP (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Prenger, R., Valle, R., Catanzaro, B.: WaveGlow: a flow-based generative network for speech synthesis. In: IEEE ICASSP (2019)","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"5_CR24","unstructured":"Qian, K., Zhang, Y., Chang, S., Yang, X., Hasegawa-Johnson, M.: AutoVC: zero-shot voice style transfer with only autoencoder loss. In: PMLR (2019)"},{"key":"5_CR25","unstructured":"Rebryk, Y., Beliaev, S.: ConVoice: Real-Time Zero-Shot Voice Style Transfer with Convolutional Network. arXiv e-prints arXiv:2005.07815 (2020)"},{"key":"5_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2014 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Shuang, Z.W., Bakis, R., Shechtman, S., Chazan, D., Qin, Y.: Frequency warping based on mapping formant parameters. In: INTERSPEECH (2006)","DOI":"10.21437\/Interspeech.2006-588"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Sisman, B., Yamagishi, J., King, S., Li, H.: An Overview of Voice Conversion and its Challenges: From Statistical Modeling to Deep Learning. arXiv e-prints arXiv:2008.03648 (2020)","DOI":"10.1109\/TASLP.2020.3038524"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Smith, L.N.: Cyclical learning rates for training neural networks. In: IEEE WACV (2017)","DOI":"10.1109\/WACV.2017.58"},{"issue":"2","key":"5_CR30","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1109\/89.661472","volume":"6","author":"Y Stylianou","year":"1998","unstructured":"Stylianou, Y., Cappe, O., Moulines, E.: Continuous probabilistic transform for voice conversion. IEEE Trans. Speech Audio Process. 6(2), 131\u2013142 (1998)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Sun, L., Kang, S., Li, K., Meng, H.: Voice conversion using deep Bidirectional Long Short-Term Memory based Recurrent Neural Networks. In: IEEE ICASSP (2015)","DOI":"10.1109\/ICASSP.2015.7178896"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Suundermann, D., Strecha, G., Bonafonte, A., H\u00f6ge, H., Ney, H.: Evaluation of VTLN-based voice conversion for embedded speech synthesis. In: INTERSPEECH (2005)","DOI":"10.21437\/Interspeech.2005-803"},{"issue":"8","key":"5_CR33","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TASL.2007.907344","volume":"15","author":"T Toda","year":"2007","unstructured":"Toda, T., Black, A.W., Tokuda, K.: Voice conversion based on maximum-likelihood estimation of spectral parameter trajectory. IEEE Trans. Audio Speech Lang. Process. 15(8), 2222\u20132235 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"5_CR34","unstructured":"Veaux, C., Yamagishi, J., Macdonald, K.: CSTR VCTK Corpus: English Multi-speaker Corpus for CSTR Voice Cloning Toolkit (2017). http:\/\/homepages.inf.ed.ac.uk\/jyamagis\/page3\/page58\/page58.html. Accessed 1 Sep 2020"},{"key":"5_CR35","doi-asserted-by":"crossref","unstructured":"Wan, L., Wang, Q., Papir, A., Moreno, I.L.: Generalized end-to-end loss for speaker verification. In: ICASSP (2018)","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"5_CR36","unstructured":"Zhao, Y., et al.: Voice Conversion Challenge 2020: Intra-lingual semi-parallel and cross-lingual voice conversion. arXiv e-prints arXiv:2008.12527 (2020)"},{"key":"5_CR37","doi-asserted-by":"publisher","first-page":"e17","DOI":"10.1017\/ATSIP.2014.17","volume":"3","author":"W Zhizheng","year":"2014","unstructured":"Zhizheng, W., Haizhou, L.: Voice conversion versus speaker verification: an overview. APSIPA Trans. Sig. Inf. Process. 3, e17 (2014)","journal-title":"APSIPA Trans. Sig. Inf. Process."}],"container-title":["Communications in Computer and Information Science","Artificial Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-66151-9_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,7]],"date-time":"2022-12-07T22:07:36Z","timestamp":1670450856000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-66151-9_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030661502","9783030661519"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-66151-9_5","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"21 December 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SACAIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Southern African Conference for Artificial Intelligence Research","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Muldersdrift","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"South Africa","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 February 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 February 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"sacair2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/sacair.org.za\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"53","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"19","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"36% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Dur to the COVID-19 pandemic SACAIR 2020 was postponed to February 2021","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}