{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T02:12:00Z","timestamp":1784945520469,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Central Government Guides Local Science and Technology Development Special Funds","award":["[2018]4008"],"award-info":[{"award-number":["[2018]4008"]}]},{"DOI":"10.13039\/501100018555","name":"Science and Technology Program of Guizhou Province","doi-asserted-by":"publisher","award":["[2023]YB449"],"award-info":[{"award-number":["[2023]YB449"]}],"id":[{"id":"10.13039\/501100018555","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s00034-025-03198-3","type":"journal-article","created":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T18:38:10Z","timestamp":1751395090000},"page":"8489-8509","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["IAFF-VC: Any-to-Any Voice Conversion Using Attentional Feature Fusion"],"prefix":"10.1007","volume":"44","author":[{"given":"Kai","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sicong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,1]]},"reference":[{"key":"3198_CR1","unstructured":"V.\u00a0D.\u00a0O. Aaron, V. Oriol et\u00a0al., Neural discrete representation learning, in Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"3198_CR2","unstructured":"V.\u00a0D.\u00a0O. Aaron, L. Yazhe, V. Oriol, Representation learning with contrastive predictive coding. arXiv:1807.03748 (2018)"},{"key":"3198_CR3","first-page":"12449","volume":"33","author":"B Alexei","year":"2020","unstructured":"B. Alexei, Z. Yuhao, M. Abdelrahman et al., wav2vec 2.0: a framework for self-supervised learning of speech representations. Adv. Neural Inf. Process. Syst. 33, 12449\u201312460 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3198_CR4","unstructured":"L. Bei, C. Zhengyang, Q. Yanmin, Attentive feature fusion for robust speaker verification, in Proc. Interspeech, pp. 286\u2013290 (2022a)"},{"key":"3198_CR5","doi-asserted-by":"publisher","first-page":"1825","DOI":"10.1109\/TASLP.2023.3273417","volume":"31","author":"L Bei","year":"2023","unstructured":"L. Bei, C. Zhengyang, Q. Yanmin, Depth-first neural architecture with attentive feature fusion for efficient speaker verification. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 1825\u20131838 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3198_CR6","unstructured":"D. Brecht, T. Jenthe, D. Kris, ECAPA-TDNN: emphasized channel attention, propagation and aggregation in TDNN based speaker verification. Interspeech (2020)"},{"key":"3198_CR7","first-page":"8284","volume":"45","author":"L Chengyang","year":"2023","unstructured":"L. Chengyang, Z. Heng, L. Yang et al., Detection-friendly dehazing: object detection in real-world hazy scenes. IEEE Trans. Pattern Anal. Mach. Intell. 45, 8284\u20138295 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3198_CR8","doi-asserted-by":"publisher","unstructured":"V. Christophe, Y. Junichi, M. Kirsten, CSTR VCTK Corpus: English Multi-speaker Corpus for CSTR Voice Cloning Toolkit. [Sound Dataset], https:\/\/doi.org\/10.7488\/ds\/1994 (2017)","DOI":"10.7488\/ds\/1994"},{"key":"3198_CR9","doi-asserted-by":"crossref","unstructured":"W. Da-Yi, L. Hung-yi, One-shot voice conversion by vector quantization, in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 7734\u20137738 (2020)","DOI":"10.1109\/ICASSP40776.2020.9053854"},{"key":"3198_CR10","unstructured":"W. Da-Yi, C. Yen-Hao, L. Hung-yi, Vqvc+: one-shot voice conversion by vector quantization and u-net architecture, in Proc. Interspeech 2020, pp. 4691\u20134695 (2020)"},{"key":"3198_CR11","doi-asserted-by":"crossref","unstructured":"O. Daliang, H. Su, Z. Guozhong et\u00a0al., Efficient multi-scale attention module with cross-spatial learning, in ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 1\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096298"},{"key":"3198_CR12","doi-asserted-by":"crossref","unstructured":"S. David, G.-R. Daniel, S. Gregory et\u00a0al., X-vectors: robust DNN embeddings for speaker recognition, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 5329\u20135333 (2018)","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"3198_CR13","first-page":"713","volume":"105\u2013D","author":"L Dichao","year":"2022","unstructured":"L. Dichao, W. Yu, M. Kenji et al., Recursive multi-scale channel-spatial attention for fine-grained image classification. IEICE Trans. Inf. Syst. 105\u2013D, 713\u2013726 (2022)","journal-title":"IEICE Trans. Inf. Syst."},{"key":"3198_CR14","unstructured":"K. Diederik\u00a0P, W. Max et\u00a0al., Auto-encoding variational bayes (2013)"},{"key":"3198_CR15","unstructured":"W. Disong, D. Liqun, Y. Yu\u00a0Ting et\u00a0al., Vqmivc: vector quantization and mutual information-based unsupervised speech representation disentanglement for one-shot voice conversion, in Proc. Interspeech 2021, pp. 1344\u20131348 (2021)"},{"key":"3198_CR16","unstructured":"U. Dmitry, V. Andrea, L. Victor, Instance normalization: the missing ingredient for fast stylization. arXiv:1607.08022 (2016)"},{"key":"3198_CR17","unstructured":"L. Drew, S. Dan, E. Sven et\u00a0al., Learning what and where to attend. arXiv:1805.08819 (2018)"},{"key":"3198_CR18","unstructured":"M. Gabriel, N. Babak, C. Assmaa et\u00a0al., Nisqa: a deep CNN-self-attention model for multidimensional speech quality prediction with crowdsourced datasets, in Proc. Interspeech 2021, pp. 2127\u20132131 (2021)"},{"key":"3198_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/09540091.2023.2257399","volume":"35","author":"W Hao","year":"2023","unstructured":"W. Hao, H. Dezhi, C. Mingming et al., Nas-yolox: a SAR ship detection using neural architecture search and multi-scale attention. Connect. Sci. 35, 1\u201332 (2023)","journal-title":"Connect. Sci."},{"key":"3198_CR20","doi-asserted-by":"crossref","unstructured":"T. Hideyuki, U. Katsuya, A. Shunsuke, Efficiently trainable text-to-speech system based on deep convolutional networks with guided attention, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 4784\u20134788 (2018)","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"3198_CR21","doi-asserted-by":"crossref","unstructured":"S. Hubert, D. Piotr, van R. Pol et\u00a0al., Wavthruvec: latent speech representation as intermediate features for neural speech synthesis, in Proc. Interspeech 2022, pp. 833\u2013837 (2022)","DOI":"10.21437\/Interspeech.2022-10797"},{"key":"3198_CR22","doi-asserted-by":"crossref","unstructured":"P. Hyun\u00a0Joon, Y. Seok\u00a0Woo, K. Jin\u00a0Sob et\u00a0al., Triaan-vc: triple adaptive attention normalization for any-to-any voice conversion, in ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 1\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096642"},{"issue":"11","key":"3198_CR23","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"G Ian","year":"2020","unstructured":"G. Ian, P.-A. Jean, M. Mehdi et al., Generative adversarial networks. Commun. ACM 63(11), 139\u2013144 (2020)","journal-title":"Commun. ACM"},{"key":"3198_CR24","unstructured":"L. Jheng-hao, L. Yist\u00a0Y, C. Chung-Ming, et\u00a0al., S2vc: a framework for any-to-any voice conversion with self-supervised pretrained representations, in Proc. Interspeech 2021, pp. 836\u2013840 (2021a)"},{"key":"3198_CR25","doi-asserted-by":"crossref","unstructured":"Q. Jiajun, G. Wu, G. Bin, Bidirectional multiscale feature aggregation for speaker verification, in Proc. Interspeech 2021, pp. 71\u201375 (2021)","DOI":"10.21437\/Interspeech.2021-111"},{"key":"3198_CR26","unstructured":"H. Jie, S. Li, S. Gang, Squeeze-and-excitation networks, in Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018a)"},{"key":"3198_CR27","unstructured":"H. Jie, S. Li, A. Samuel et\u00a0al., Gather-excite: exploiting feature context in convolutional neural networks, in Advances in Neural Information Processing Systems, vol. 31 (2018b)"},{"key":"3198_CR28","doi-asserted-by":"crossref","unstructured":"S. Jonathan, P. Ruoming, W. Ron\u00a0J et\u00a0al., Natural tts synthesis by conditioning wavenet on mel spectrogram predictions, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 4779\u20134783 (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"3198_CR29","unstructured":"C. Ju-chieh, L. Hung-Yi, One-shot voice conversion by separating speaker and content representations with instance normalization, in Proc. Interspeech 2019, pp. 664\u2013668 (2019)"},{"key":"3198_CR30","first-page":"17022","volume":"33","author":"K Jungil","year":"2020","unstructured":"K. Jungil, K. Jaehyeon, B. Jaekyoung, Hifi-gan: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural Inf. Process. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3198_CR31","unstructured":"Q. Kaizhi, Z. Yang, C. Shiyu et\u00a0al., Autovc: zero-shot voice style transfer with only autoencoder loss, in International Conference on Machine Learning, PMLR, pp. 5210\u20135219 (2019)"},{"key":"3198_CR32","unstructured":"R. Morgane, J. Armand, M. Pierre-Emmanuel et\u00a0al., Unsupervised pretraining transfers well across languages, in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 7414\u20137418 (2020)"},{"key":"3198_CR33","unstructured":"H. Qibin, Z. Daquan, F. Jiashi, Coordinate attention for efficient mobile network design, in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13713\u201313722 (2021)"},{"key":"3198_CR34","unstructured":"Y. Ryuichi, S. Eunwoo, K. Jae-Min, Parallel wavegan: a fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram, in ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 6199\u20136203 (2020)"},{"key":"3198_CR35","unstructured":"D. Sander, Z. Heiga, S. Karen et\u00a0al., Wavenet: a generative model for raw audio. arXiv:1609.03499 12 (2016)"},{"key":"3198_CR36","doi-asserted-by":"crossref","unstructured":"W. Sanghyun, P. Jongchan, L. Joon-Young et\u00a0al., Cbam: convolutional block attention module, in Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"issue":"6","key":"3198_CR37","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"C Sanyuan","year":"2022","unstructured":"C. Sanyuan, W. Chengyi, C. Zhengyang et al., WAVLM: large-scale self-supervised pre-training for full stack speech processing. IEEE J. Select. Top. Signal Process. 16(6), 1505\u20131518 (2022)","journal-title":"IEEE J. Select. Top. Signal Process."},{"key":"3198_CR38","unstructured":"H. Wei-Ning, T. Yao-Hung\u00a0Hubert, B. Benjamin et\u00a0al., Hubert: how much can a bad teacher benefit ASR pre-training? in ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 6533\u20136537 (2021)"},{"key":"3198_CR39","unstructured":"C. William, J. Navdeep, L. Quoc et\u00a0al., Listen, attend and spell: a neural network for large vocabulary conversational speech recognition, in 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 4960\u20134964 (2016)"},{"key":"3198_CR40","doi-asserted-by":"publisher","first-page":"3431","DOI":"10.1109\/TASLP.2023.3306716","volume":"31","author":"X Xuexin","year":"2023","unstructured":"X. Xuexin, S. Liang, C. Xun-Yu et al., Any-to-any voice conversion with multi-layer speaker adaptation and content supervision. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 3431\u20133445 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3198_CR41","unstructured":"C. Yafeng, Z. Siqi, W. Hui et\u00a0al., An enhanced res2net with local and global feature fusion for speaker verification. arXiv:2305.12838 (2023)"},{"key":"3198_CR42","unstructured":"C. Yen-Hao, W. Da-Yi, W. Tsung-Han et\u00a0al., Again-vc: a one-shot voice conversion using activation guidance and adaptive instance normalization, in ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 5954\u20135958 (2021)"},{"key":"3198_CR43","unstructured":"G. Yewei, Z. Zhenyu, Y. Xiaowei et\u00a0al., Mediumvc: any-to-any voice conversion using synthetic specific-speaker speeches as intermedium features. arXiv:2110.02500 (2021)"},{"key":"3198_CR44","doi-asserted-by":"crossref","unstructured":"Z. Yi, H. Wen-Chin, T. Xiaohai et\u00a0al., Voice conversion challenge 2020: intra-lingual semi-parallel and cross-lingual voice conversion. arXiv:2008.12527 (2020)","DOI":"10.21437\/VCCBC.2020-14"},{"key":"3198_CR45","doi-asserted-by":"crossref","unstructured":"D. Yimian, G. Fabian, O. Stefan et\u00a0al., Attentional feature fusion, in 2021 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 3559\u20133568. https:\/\/api.semanticscholar.org\/CorpusID:221995547 (2020)","DOI":"10.1109\/WACV48630.2021.00360"},{"key":"3198_CR46","doi-asserted-by":"crossref","unstructured":"Y.Y. Lin et\u00a0al., Fragmentvc: any-to-any voice conversion by end-to-end extracting and fusing fine-grained voice fragments with attention, in ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), IEEE, pp. 5939\u20135943 (2021b)","DOI":"10.1109\/ICASSP39728.2021.9413699"},{"key":"3198_CR47","unstructured":"W. Yuxuan, S.-R. RJ, S. Daisy et\u00a0al., Tacotron: towards end-to-end speech synthesis, in Proc. Interspeech 2017, pp. 4006\u20134010 (2017)"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03198-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-025-03198-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03198-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T06:13:34Z","timestamp":1761977614000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-025-03198-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,1]]},"references-count":47,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["3198"],"URL":"https:\/\/doi.org\/10.1007\/s00034-025-03198-3","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,1]]},"assertion":[{"value":"3 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 May 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 May 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no Conflict of interest to declare.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}