{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T23:58:39Z","timestamp":1774742319408,"version":"3.50.1"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2024,4,21]],"date-time":"2024-04-21T00:00:00Z","timestamp":1713657600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,4,21]],"date-time":"2024-04-21T00:00:00Z","timestamp":1713657600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Project","award":["No.2020AAA0106200"],"award-info":[{"award-number":["No.2020AAA0106200"]}]},{"name":"the National Nature Science Foundation of China under Grants","award":["No.61872199"],"award-info":[{"award-number":["No.61872199"]}]},{"name":"the National Nature Science Foundation of China under Grants","award":["No.61872424"],"award-info":[{"award-number":["No.61872424"]}]},{"name":"the National Nature Science Foundation of China under Grants","award":["No.61936005"],"award-info":[{"award-number":["No.61936005"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2024,7]]},"DOI":"10.1007\/s00034-024-02675-5","type":"journal-article","created":{"date-parts":[[2024,4,21]],"date-time":"2024-04-21T14:01:21Z","timestamp":1713708081000},"page":"4565-4587","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["One-Shot Voice Conversion Based on Style Generative Adversarial Networks with ESR and DSNet"],"prefix":"10.1007","volume":"43","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6208-556X","authenticated-orcid":false,"given":"Yanping","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangtian","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeyu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhicheng","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,4,21]]},"reference":[{"key":"2675_CR1","doi-asserted-by":"crossref","unstructured":"M. Baas, H. Kamper, Stargan-zsvc: Towards zero-shot voice conversion in low-resource contexts. In: Southern African conference for artificial intelligence research, pp 69\u201384 (2020)","DOI":"10.1007\/978-3-030-66151-9_5"},{"key":"2675_CR2","doi-asserted-by":"crossref","unstructured":"M. Chen, Y. Shi, T. Hain, Towards low-resource stargan voice conversion using weight adaptive instance normalization. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5949\u20135953 (2021)","DOI":"10.1109\/ICASSP39728.2021.9415042"},{"key":"2675_CR3","doi-asserted-by":"crossref","unstructured":"C. Chun, Y.H. Lee, G.W. Lee, et\u00a0al., (2023) Non-parallel voice conversion using cycle-consistent adversarial networks with self-supervised representations. In: 2023 IEEE 20th consumer communications & networking conference (CCNC), pp 931\u2013932","DOI":"10.1109\/CCNC51644.2023.10060510"},{"key":"2675_CR4","unstructured":"S. Ghosh, Y. Sinha, I. Siegert, et\u00a0al., (2023) Improving voice conversion for dissimilar speakers using perceptual losses. Deutsche Gesellschaft f\u00fcr Akustik eV pp 1358\u20131361"},{"key":"2675_CR5","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, et\u00a0al., Deep residual learning for image recognition. in Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2675_CR6","doi-asserted-by":"crossref","unstructured":"C.C. Hsu, H.T. Hwang, Y.C. Wu, et\u00a0al., Voice conversion from unaligned corpora using variational autoencoding wasserstein generative adversarial networks. arXiv preprint arXiv:1704.00849 (2017)","DOI":"10.21437\/Interspeech.2017-63"},{"key":"2675_CR7","doi-asserted-by":"crossref","unstructured":"G. Huang, Z. Liu, L. Van Der\u00a0Maaten et\u00a0al., Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4700\u20134708 (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"2675_CR8","doi-asserted-by":"crossref","unstructured":"X. Huang, S. Belongie, Arbitrary style transfer in real-time with adaptive instance normalization. in Proceedings of the IEEE International Conference on Computer Vision, pp 1501\u20131510 (2017)","DOI":"10.1109\/ICCV.2017.167"},{"key":"2675_CR9","unstructured":"S. Ioffe, C. Szegedy, Batch normalization: Accelerating deep network training by reducing internal covariate shift, in International Conference on Machine Learning, pp 448\u2013456 (2015)"},{"key":"2675_CR10","doi-asserted-by":"crossref","unstructured":"H. Kameoka, T. Kaneko, K. Tanaka, et\u00a0al., Stargan-vc: Non-parallel many-to-many voice conversion using star generative adversarial networks. in: 2018 IEEE spoken language technology workshop (SLT), pp 266\u2013273 (2018)","DOI":"10.1109\/SLT.2018.8639535"},{"key":"2675_CR11","doi-asserted-by":"publisher","first-page":"2982","DOI":"10.1109\/TASLP.2020.3036784","volume":"28","author":"H Kameoka","year":"2020","unstructured":"H. Kameoka, T. Kaneko, K. Tanaka et al., Nonparallel voice conversion with augmented classifier star generative adversarial networks. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 28, 2982\u20132995 (2020)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"2675_CR12","doi-asserted-by":"crossref","unstructured":"T. Kaneko, H. Kameoka, Parallel-data-free voice conversion using cycle-consistent adversarial networks. arXiv preprint arXiv:1711.11293 (2017)","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"2675_CR13","doi-asserted-by":"crossref","unstructured":"T. Kaneko, H. Kameoka, K. Tanaka, et\u00a0al., Stargan-vc2: Rethinking conditional methods for stargan-based voice conversion. arXiv preprint arXiv:1907.12279 (2019)","DOI":"10.21437\/Interspeech.2019-2236"},{"key":"2675_CR14","doi-asserted-by":"crossref","unstructured":"T. Karras, S. Laine, M. Aittala, et\u00a0al., Analyzing and improving the image quality of stylegan. in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 8110\u20138119 (2020)","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"2675_CR15","unstructured":"D. P. Kingma, J. Ba, Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"2675_CR16","doi-asserted-by":"crossref","unstructured":"Y. Li, D. Xu, Y. Zhang, et\u00a0al., Non-parallel many-to-many voice conversion with psr-stargan. In: Interspeech, pp 781\u2013785 (2020)","DOI":"10.21437\/Interspeech.2020-1310"},{"issue":"10","key":"2675_CR17","doi-asserted-by":"publisher","first-page":"2150188","DOI":"10.1142\/S0218126621501887","volume":"30","author":"Y Li","year":"2021","unstructured":"Y. Li, Z. He, Y. Zhang et al., High-quality many-to-many voice conversion using transitive star generative adversarial networks with adaptive instance normalization. J. Circuits Syst. Comput. 30(10), 2150188 (2021)","journal-title":"J. Circuits Syst. Comput."},{"issue":"8","key":"2675_CR18","doi-asserted-by":"publisher","first-page":"4632","DOI":"10.1007\/s00034-022-01998-5","volume":"41","author":"Y Li","year":"2022","unstructured":"Y. Li, X. Qiu, P. Cao et al., Non-parallel voice conversion based on perceptual star generative adversarial network. Circuits Syst. Signal Process. 41(8), 4632\u20134648 (2022)","journal-title":"Circuits Syst. Signal Process."},{"key":"2675_CR19","doi-asserted-by":"crossref","unstructured":"B. Lim, S. Son, H. Kim, et\u00a0al., (2017) Enhanced deep residual networks for single image super-resolution. in Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition workshops, pp 136\u2013144","DOI":"10.1109\/CVPRW.2017.151"},{"key":"2675_CR20","doi-asserted-by":"crossref","unstructured":"K. Liu, J. Zhang, Y. Yan, High quality voice conversion through phoneme-based linear mapping functions with straight for mandarin. in Fourth international conference on fuzzy systems and knowledge discovery (FSKD), pp 410\u2013414 (2007)","DOI":"10.1109\/FSKD.2007.347"},{"key":"2675_CR21","doi-asserted-by":"crossref","unstructured":"J. Lorenzo-Trueba, J. Yamagishi, T. Toda, et\u00a0al., The voice conversion challenge 2018: Promoting development of parallel and nonparallel methods. arXiv preprint arXiv:1804.04262 (2018)","DOI":"10.21437\/Odyssey.2018-28"},{"key":"2675_CR22","doi-asserted-by":"publisher","first-page":"2967","DOI":"10.1109\/TASLP.2020.3034994","volume":"28","author":"HT Luong","year":"2020","unstructured":"H.T. Luong, J. Yamagishi, Nautilus: a versatile voice cloning system. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 28, 2967\u20132981 (2020)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"issue":"7","key":"2675_CR23","doi-asserted-by":"publisher","first-page":"1877","DOI":"10.1587\/transinf.2015EDP7457","volume":"99","author":"M Morise","year":"2016","unstructured":"M. Morise, F. Yokomori, K. Ozawa, World: a vocoder-based high-quality speech synthesis system for real-time applications. IEICE Trans. Inf. Syst. 99(7), 1877\u20131884 (2016)","journal-title":"IEICE Trans. Inf. Syst."},{"key":"2675_CR24","doi-asserted-by":"crossref","unstructured":"X. Qiu, Y. Luo, Research on synthesis of designated speaker speech based on stargan-vc model. in International Conference on Artificial Intelligence and Intelligent Information Processing (AIIIP), pp 256\u2013263 (2022)","DOI":"10.1117\/12.2659719"},{"key":"2675_CR25","doi-asserted-by":"crossref","unstructured":"S. Sakamoto, A. Taniguchi, T. Taniguchi, et\u00a0al., Stargan-vc+ asr: Stargan-based non-parallel voice conversion regularized by automatic speech recognition. arXiv preprint arXiv:2108.04395 (2021)","DOI":"10.21437\/Interspeech.2021-492"},{"key":"2675_CR26","doi-asserted-by":"crossref","unstructured":"S. Si, J. Wang, X. Zhang, et\u00a0al., Boosting stargans for voice conversion with contrastive discriminator. in International Conference on Neural Information Processing, pp 355\u2013366 (2022)","DOI":"10.1007\/978-3-031-30108-7_30"},{"key":"2675_CR27","doi-asserted-by":"crossref","unstructured":"D. Wang, J. Yu, X. Wu, et\u00a0al., End-to-end voice conversion via cross-modal knowledge distillation for dysarthric speech reconstruction. in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp 7744\u20137748 (2020a)","DOI":"10.1109\/ICASSP40776.2020.9054596"},{"key":"2675_CR28","doi-asserted-by":"crossref","unstructured":"R. Wang, Y. Ding, L. Li, et\u00a0al. One-shot voice conversion using star-gan. in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp 7729\u20137733 (2020b)","DOI":"10.1109\/ICASSP40776.2020.9053842"},{"key":"2675_CR29","unstructured":"Y. Wang, D. Stanton, Y. Zhang, et\u00a0al., Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis, in International Conference on Machine Learning, pp 5180\u20135189 (2018)"},{"key":"2675_CR30","doi-asserted-by":"crossref","unstructured":"M. Zhang, X. Wang, F. Fang, et\u00a0al., Joint training framework for text-to-speech and voice conversion using multi-source tacotron and wavenet. arXiv preprint arXiv:1903.12389 (2019)","DOI":"10.21437\/Interspeech.2019-1357"},{"key":"2675_CR31","doi-asserted-by":"crossref","unstructured":"S. Zhao, T.H. Nguyen, H. Wang, et\u00a0al., Fast learning for non-parallel many-to-many voice conversion with residual star generative adversarial networks. In: Interspeech, pp 689\u2013693 (2019)","DOI":"10.21437\/Interspeech.2019-2067"},{"key":"2675_CR32","unstructured":"Y. Zhao, W.C. Huang, X. Tian, et\u00a0al., Voice conversion challenge 2020: Intra-lingual semi-parallel and cross-lingual voice conversion. arXiv preprint arXiv:2008.12527 (2020)"},{"key":"2675_CR33","doi-asserted-by":"publisher","first-page":"4613","DOI":"10.1109\/TNSRE.2023.3331524","volume":"31","author":"WZ Zheng","year":"2023","unstructured":"W.Z. Zheng, J.Y. Han, C.Y. Chen et al., Improving the efficiency of dysarthria voice conversion system based on data augmentation. IEEE Trans. Neural Syst. Rehabil. Eng. 31, 4613\u20134623 (2023)","journal-title":"IEEE Trans. Neural Syst. Rehabil. Eng."}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02675-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-024-02675-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-024-02675-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,16]],"date-time":"2024-07-16T11:16:12Z","timestamp":1721128572000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-024-02675-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,21]]},"references-count":33,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2024,7]]}},"alternative-id":["2675"],"URL":"https:\/\/doi.org\/10.1007\/s00034-024-02675-5","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4,21]]},"assertion":[{"value":"9 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 March 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 March 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 April 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}