{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T15:40:10Z","timestamp":1745595610851,"version":"3.40.4"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819658800","type":"print"},{"value":"9789819658817","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-5881-7_22","type":"book-chapter","created":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T15:10:54Z","timestamp":1745593854000},"page":"284-295","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["RE-VC: Robust Zero-Shot Voice Conversion Model for\u00a0Realistic Environments"],"prefix":"10.1007","author":[{"given":"Luong","family":"Ho","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Do","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minh","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Duc","family":"Chau","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,17]]},"reference":[{"key":"22_CR1","doi-asserted-by":"crossref","unstructured":"Chen, S., et al.: Wavlm: large-scale self-supervised pre-training for full stack speech processing. IEEE J. Sel. Top. Signal Process. 16, 1505\u20131518 (2021). https:\/\/api.semanticscholar.org\/CorpusID:239885872","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"22_CR2","doi-asserted-by":"crossref","unstructured":"Chen, Y.H., Wu, D.Y., Wu, T.H., yi\u00a0Lee, H.: Again-VC: a one-shot voice conversion using activation guidance and adaptive instance normalization. In: ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5954\u20135958 (2020). https:\/\/api.semanticscholar.org\/CorpusID:226226852","DOI":"10.1109\/ICASSP39728.2021.9414257"},{"key":"22_CR3","doi-asserted-by":"crossref","unstructured":"Choi, H., Lee, S.H., Lee, S.W.: DDDM-VC: decoupled denoising diffusion models with disentangled representation and prior mixup for verified robust voice conversion. In: AAAI Conference on Artificial Intelligence (2023). https:\/\/api.semanticscholar.org\/CorpusID:258887678","DOI":"10.1609\/aaai.v38i16.29740"},{"key":"22_CR4","unstructured":"Choi, H.S., Lee, J., Kim, W.S., Lee, J.H., Heo, H., Lee, K.: Neural analysis and synthesis: Reconstructing speech from self-supervised representations. arXiv:abs\/2110.14513 (2021). https:\/\/api.semanticscholar.org\/CorpusID:239998228"},{"key":"22_CR5","doi-asserted-by":"crossref","unstructured":"Chou, J.C., Chieh Yeh, C., yi\u00a0Lee, H.: One-shot voice conversion by separating speaker and content representations with instance normalization. arXiv:abs\/1904.05742 (2019). https:\/\/api.semanticscholar.org\/CorpusID:119304586","DOI":"10.21437\/Interspeech.2019-2663"},{"key":"22_CR6","doi-asserted-by":"crossref","unstructured":"Cong, B.N., Cardinaux, F.: NVC-Net: End-to-end adversarial voice conversion. In: ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7012\u20137016 (2021). https:\/\/api.semanticscholar.org\/CorpusID:235294129","DOI":"10.1109\/ICASSP43922.2022.9747020"},{"key":"22_CR7","doi-asserted-by":"crossref","unstructured":"Hsu, C.C., Hwang, H.T., Wu, Y.C., Tsao, Y., Wang, H.: Voice conversion from unaligned corpora using variational autoencoding wasserstein generative adversarial networks. arXiv:abs\/1704.00849 (2017). https:\/\/api.semanticscholar.org\/CorpusID:1346276","DOI":"10.21437\/Interspeech.2017-63"},{"key":"22_CR8","doi-asserted-by":"crossref","unstructured":"Kameoka, H., Kaneko, T., Tanaka, K., Hojo, N.: Stargan-VC: non-parallel many-to-many voice conversion using star generative adversarial networks. In: 2018 IEEE Spoken Language Technology Workshop (SLT) pp. 266\u2013273 (2018). https:\/\/api.semanticscholar.org\/CorpusID:46949741","DOI":"10.1109\/SLT.2018.8639535"},{"key":"22_CR9","doi-asserted-by":"crossref","unstructured":"Kaneko, T., Kameoka, H., Tanaka, K., Hojo, N.: Cyclegan-VC3: examining and improving cyclegan-VCS for mel-spectrogram conversion. In: Interspeech (2020). https:\/\/api.semanticscholar.org\/CorpusID:225040879","DOI":"10.21437\/Interspeech.2020-2280"},{"key":"22_CR10","unstructured":"Kim, J., Kim, S., Kong, J., Yoon, S.: Glow-TTS: a generative flow for text-to-speech via monotonic alignment search. arXiv:abs\/2005.11129 (2020). https:\/\/api.semanticscholar.org\/CorpusID:218862956"},{"key":"22_CR11","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. arXiv:abs\/2106.06103 (2021). https:\/\/api.semanticscholar.org\/CorpusID:235417304"},{"key":"22_CR12","doi-asserted-by":"publisher","unstructured":"Li, J., Tu, W., Xiao, L.: FreeVC: towards high-quality text-free one-shot voice conversion. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10095191","DOI":"10.1109\/ICASSP49357.2023.10095191"},{"key":"22_CR13","doi-asserted-by":"crossref","unstructured":"Li, T., Liu, Y., Hu, C., Zhao, H.: CVC: contrastive learning for non-parallel voice conversion. In: Interspeech (2020). https:\/\/api.semanticscholar.org\/CorpusID:226227472","DOI":"10.21437\/Interspeech.2021-137"},{"key":"22_CR14","doi-asserted-by":"crossref","unstructured":"Li, Y.A., Han, C., Mesgarani, N.: Styletts-VC: one-shot voice conversion by knowledge transfer from style-based TTS models. In: 2022 IEEE Spoken Language Technology Workshop (SLT), pp. 920\u2013927 (2022). https:\/\/api.semanticscholar.org\/CorpusID:255341157","DOI":"10.1109\/SLT54892.2023.10022498"},{"key":"22_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y.A., Zare, A.A., Mesgarani, N.: Starganv2-VC: a diverse, unsupervised, non-parallel framework for natural-sounding voice conversion. In: Interspeech (2021). https:\/\/api.semanticscholar.org\/CorpusID:236170960","DOI":"10.21437\/Interspeech.2021-319"},{"key":"22_CR16","doi-asserted-by":"crossref","unstructured":"Hao Lin, J., Lin, Y.Y., Chien, C.M., Yi\u00a0Lee, H.: S2VC: a framework for any-to-any voice conversion with self-supervised pretrained representations. arXiv:abs\/2104.02901 (2021). https:\/\/api.semanticscholar.org\/CorpusID:233169061","DOI":"10.21437\/Interspeech.2021-1356"},{"key":"22_CR17","doi-asserted-by":"crossref","unstructured":"Mao, X., Li, Q., Xie, H., Lau, R.Y.K., Wang, Z., Smolley, S.P.: Least squares generative adversarial networks. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 2813\u20132821 (2016). https:\/\/api.semanticscholar.org\/CorpusID:206771128","DOI":"10.1109\/ICCV.2017.304"},{"key":"22_CR18","unstructured":"Min, D., Lee, D.B., Yang, E., Hwang, S.J.: Meta-stylespeech: multi-speaker adaptive text-to-speech generation. arXiv:abs\/2106.03153 (2021). https:\/\/api.semanticscholar.org\/CorpusID:235359041"},{"key":"22_CR19","doi-asserted-by":"crossref","unstructured":"Park, H.J., Yang, S.W., Kim, J.S., Shin, W., Han, S.W.: Triaan-VC: triple adaptive attention normalization for any-to-any voice conversion. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257557422","DOI":"10.1109\/ICASSP49357.2023.10096642"},{"key":"22_CR20","unstructured":"Popov, V., Vovk, I., Gogoryan, V., Sadekova, T., Kudinov, M.S., Wei, J.: Diffusion-based voice conversion with fast maximum likelihood sampling scheme. In: International Conference on Learning Representations (2022), https:\/\/openreview.net\/forum?id=8c50f-DoWAu"},{"key":"22_CR21","doi-asserted-by":"crossref","unstructured":"Prenger, R.J., Valle, R., Catanzaro, B.: Waveglow: a flow-based generative network for speech synthesis. In: ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 3617\u20133621 (2018). https:\/\/api.semanticscholar.org\/CorpusID:53145796","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"22_CR22","unstructured":"Qian, K., Zhang, Y., Chang, S., Yang, X., Hasegawa-Johnson, M.A.: Zero-shot voice style transfer with only autoencoder loss. arXiv:abs\/1905.05879 (2019). https:\/\/api.semanticscholar.org\/CorpusID:155091770"},{"key":"22_CR23","doi-asserted-by":"crossref","unstructured":"Tang, H., Zhang, X., Wang, J., Cheng, N., Xiao, J.: AVQVC: one-shot voice conversion by vector quantization with applying contrastive learning. In: ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4613\u20134617 (2022). https:\/\/api.semanticscholar.org\/CorpusID:247011993","DOI":"10.1109\/ICASSP43922.2022.9746369"},{"key":"22_CR24","doi-asserted-by":"crossref","unstructured":"Wang, D., Deng, L., Yeung, Y.T., Chen, X., Liu, X., Meng, H.M.: VQMIVC: vector quantization and mutual information-based unsupervised speech representation disentanglement for one-shot voice conversion. arXiv:abs\/2106.10132 (2021). https:\/\/api.semanticscholar.org\/CorpusID:235485208","DOI":"10.21437\/Interspeech.2021-283"},{"key":"22_CR25","doi-asserted-by":"publisher","unstructured":"Wang, H., et al.: DuTa-VC: a duration-aware typical-to-atypical voice conversion approach with diffusion probabilistic model. In: Proceedings of the INTERSPEECH 2023, pp. 1548\u20131552 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-2203","DOI":"10.21437\/Interspeech.2023-2203"},{"key":"22_CR26","doi-asserted-by":"crossref","unstructured":"Wu, D.Y., Chen, Y.H., Yi\u00a0Lee, H.: VQVC+: one-shot voice conversion by vector quantization and u-net architecture. arXiv:abs\/2006.04154 (2020). https:\/\/api.semanticscholar.org\/CorpusID:219531044","DOI":"10.21437\/Interspeech.2020-1443"},{"key":"22_CR27","doi-asserted-by":"publisher","unstructured":"Xu, L., Zhong, R., Liu, Y., Yang, H., Zhang, S.: Flow-VAE VC: end-to-end flow framework with contrastive loss for zero-shot voice conversion. In: Proceedings of the INTERSPEECH 2023, pp. 2293\u20132297 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-1508","DOI":"10.21437\/Interspeech.2023-1508"},{"key":"22_CR28","unstructured":"Yamagishi, J., Veaux, C., MacDonald, K.: CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92) (2019). https:\/\/api.semanticscholar.org\/CorpusID:213060286"},{"key":"22_CR29","doi-asserted-by":"crossref","unstructured":"Zen, H., et al.: LibriTTS: a corpus derived from LibriSpeech for text-to-speech. In: Interspeech (2019). https:\/\/api.semanticscholar.org\/CorpusID:102352475","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"22_CR30","unstructured":"Zhang, B., et al.: WeNet: production first and production ready end-to-end speech recognition toolkit. arXiv:abs\/2102.01547 (2021). https:\/\/api.semanticscholar.org\/CorpusID:231749822"}],"container-title":["Communications in Computer and Information Science","Recent Challenges in Intelligent Information and Database Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-5881-7_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T15:11:11Z","timestamp":1745593871000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-5881-7_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819658800","9789819658817"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-5881-7_22","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"17 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ACIIDS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Intelligent Information and Database Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kitakyushu","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aciids2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/aciids.pwr.edu.pl\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}