{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,4]],"date-time":"2026-02-04T05:21:41Z","timestamp":1770182501870,"version":"3.49.0"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100020950","name":"National Science and Technology Council","doi-asserted-by":"publisher","award":["113-2221-E-415-010"],"award-info":[{"award-number":["113-2221-E-415-010"]}],"id":[{"id":"10.13039\/501100020950","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100020950","name":"National Science and Technology Council","doi-asserted-by":"publisher","award":["113-2811-E-006-036"],"award-info":[{"award-number":["113-2811-E-006-036"]}],"id":[{"id":"10.13039\/501100020950","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-026-21200-1","type":"journal-article","created":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T16:57:28Z","timestamp":1770137848000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Effective personalized cross-lingual TTS with speech synthesis of prosodic naturalness and vivid emotion"],"prefix":"10.1007","volume":"85","author":[{"given":"Din-Yuen","family":"Chan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun-Hao","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jhing-Fa","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7158-1725","authenticated-orcid":false,"given":"Hsin-Chun","family":"Tsai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,3]]},"reference":[{"key":"21200_CR1","doi-asserted-by":"crossref","unstructured":"Adam Gabry\u015b G, Huybrechts MS, Ribeiro C-M, Chien J, Roth G, Comini R, Barra-Chicote B, Perz, and Jaime Lorenzo-Trueba (2022) Voice Filter: Few-Shot Text-to-Speech Speaker Adaptation Using Voice Conversion as a Post-Processing Module. In IEEE International Conference on Acoustics, SpeechSignal Processing (ICASSP)","DOI":"10.1109\/ICASSP43922.2022.9747239"},{"key":"21200_CR2","unstructured":"Alec Radford JW, Kim T, Xu G, Brockman C, McLeavey, Sutskever I (2023) Robust speech recognition via large-scale weak supervision. In International Conference on Machine Learning, pages 28492\u201328518"},{"key":"21200_CR3","unstructured":"Biographies of Authors"},{"key":"21200_CR4","doi-asserted-by":"crossref","unstructured":"Chen S, Wang C, Wu Y, Zhang Z, Zhou L, Liu S, Chen Z, Liu Y, Wang H, Li J, He L, Zhao S, and Furu Wei (2025) Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers. IEEE\/ACM Transactions on audio, speech,language processing, 33: pages 705\u2013717","DOI":"10.1109\/TASLPRO.2025.3530270"},{"key":"21200_CR5","unstructured":"Christophe Veaux J, Yamagishi, Kirsten, MacDonald (2017) Cstr Vctk corpus: english multi-speaker corpus for Cstr voice cloning toolkit. 41Din Yuen Chan, Jhing-Fa Wang and Hsu-Ting chin (Oct. 2023) A new speaker-diarization technology with denoising spectral-LSTM for online automatic multi-dialogue recording. Multimedia Tools and Applications"},{"key":"21200_CR6","doi-asserted-by":"crossref","unstructured":"Chung Y-A, Wang Y, Hsu W-N, Zhang Y, Skerry-Ryan RJ (2019) Semi- supervised training for improving data efficiency in end-to-end speech synthesis. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 6940\u20136944","DOI":"10.1109\/ICASSP.2019.8683862"},{"key":"21200_CR7","doi-asserted-by":"crossref","unstructured":"Daniil R\u00f5bnikov and Tanel Alum\u00e4e (2024) Single-Stage TTS with Adapted Vocoder and Cross-Attention: Taltech Systems for the Limmits\u201924 Challenge. in Proceedings of the IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops (ICASSPW)","DOI":"10.1109\/ICASSPW62465.2024.10627656"},{"key":"21200_CR8","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: Pretraining of deep bidirectional transformers for language understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis, 1, pages. 4171\u20134186","DOI":"10.18653\/v1\/N19-1423"},{"key":"21200_CR9","unstructured":"Edresson Casanova J, Weber C, Shulby AC, Junior E, G\u00f6lge, Moacir Antonelli P (2022) YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for everyone. In Proceedings of the 39th International Conference on Machine Learning, PMLR 162:2709\u20132720"},{"issue":"11","key":"21200_CR10","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Yoshua Bengio (2020) Generative adversarial networks. Commun ACM 63(11):139\u2013144","journal-title":"Commun ACM"},{"key":"21200_CR11","doi-asserted-by":"crossref","unstructured":"Guo Y, Lv Y, Dou J, Zhang Y, Wang Y (2024) FLY-TTS: Fast, Lightweight and High-Quality End-to-End Text-to-Speech Synthesis. In Interspeech","DOI":"10.21437\/Interspeech.2024-1435"},{"key":"21200_CR12","doi-asserted-by":"crossref","unstructured":"Han W, Kang M, Kim C, Yang E (2025) Stable-TTS: Stable Speaker-Adaptive Text-to-Speech Synthesis via Prosody Prompting. in Proceedings of the IEEE\/CVF international conference on Acoustics, Speech and Signal Processing (ICASSP)","DOI":"10.1109\/ICASSP49660.2025.10890553"},{"key":"21200_CR13","doi-asserted-by":"crossref","unstructured":"Heiga Zen and Ha\u00b8sim Sak (2015) Unidirectional long short-term memory recurrent neural network with recurrent output layer for low-latency speech synthesis. in Proceedings of the IEEE\/CVF International conference on Acoustics, Speech and Signal Processing (ICASSP), pages 4470\u2013 4474","DOI":"10.1109\/ICASSP.2015.7178816"},{"issue":"11","key":"21200_CR14","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"K Heiga Zen","year":"2009","unstructured":"Heiga Zen K, Tokuda, Alan W, Black (2009) Statistical parametric speech synthesis. Speech Commun 51(11):1039\u20131064","journal-title":"Speech Commun"},{"key":"21200_CR15","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W-N Hsu","year":"2021","unstructured":"Hsu W-N, Bolte B, Tsai Y-HH, Lakhotia K, Salakhutdinov R, Mohamed A (2021) Hubert: self-supervised speech representation learning by masked prediction of hidden units. IEEE ACM Trans Audio Speech Lang Process 29:3451\u20133460","journal-title":"IEEE ACM Trans Audio Speech Lang Process"},{"key":"21200_CR16","unstructured":"Hyunjae Cho W, Jung J, Lee, Hoon S (2022) Woo SANE-TTS: stable and natural end-to-end multilingual text-to-speech. arXiv preprint arXiv:2206.12132"},{"key":"21200_CR17","doi-asserted-by":"crossref","unstructured":"Im C-B, Lee S-H, Kim S-B, Seong-Whan L (2022) Emoq-tts: Emotion intensity quantization for fine-grained controllable emotional text-to-speech. In ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 6317\u20136321","DOI":"10.1109\/ICASSP43922.2022.9747098"},{"key":"21200_CR18","doi-asserted-by":"crossref","unstructured":"Keiichi Tokuda Y, Nankaku T, Toda H, Zen J, Yamagishi, Oura K (2013) Speech synthesis based on hidden markov models. Proceedings of the IEEE, 101(5):1234\u20131252","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"21200_CR19","first-page":"8067","volume":"33","author":"J Kim","year":"2020","unstructured":"Kim J, Kim S, Kong J, Yoon S (2020) Glow-tts: a generative flow for text-to-speech via monotonic alignment search. Adv Neural Inf Process Syst 33:8067\u20138077","journal-title":"Adv Neural Inf Process Syst"},{"key":"21200_CR20","unstructured":"Kim J, Kong J, Son J (2021) Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In Proceedings of the International Conference on Machine Learning (ICML), pages 5530\u20135540"},{"key":"21200_CR21","doi-asserted-by":"crossref","unstructured":"Kun Zhou B, Sisman, Li H (2021) Limited Data Emotional Voice Conversion Leveraging Text-to-Speech: Two-Stage Sequence-to-Sequence Training. in Interspeech","DOI":"10.21437\/Interspeech.2021-781"},{"key":"21200_CR22","doi-asserted-by":"crossref","unstructured":"Lee S-H, Choi H-Y, Kim S-B, Seong-Whan Lee (2025) HierSpeech++: bridging the gap between semantic and acoustic representation of speech by hierarchical variational inference for Zero-shot speech synthesis. IEEE Trans Neural Networks Learn Syst, vol. 36, no. 10 pages 1\u201315","DOI":"10.1109\/TNNLS.2025.3584944"},{"key":"21200_CR23","doi-asserted-by":"crossref","unstructured":"Li N, Liu S, Liu Y, Zhao S, Liu M (2019) Neural speech synthesis with transformer network. In Proceedings of the AAAI conference on artificial intelligence, volume 33, pages 6706\u20136713","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"21200_CR24","doi-asserted-by":"crossref","unstructured":"Li-Wei Chen and Alexander Rudnicky (2022) Fine-grained style control in transformer-based text- to-speech synthesis. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 7907\u20137911","DOI":"10.1109\/ICASSP43922.2022.9747747"},{"key":"21200_CR25","doi-asserted-by":"crossref","unstructured":"Liu Z, Mao H, Wu C-Y, Feichtenhofer C, Darrell T, Xie S (2022) A convnet for the 2020s. in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp.11 976\u2009\u2013\u200911 986","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"21200_CR26","unstructured":"Nal Kalchbrenner E, Elsen K, Simonyan S, Noury N, Casagrande E, Lockhart F, Stimberg (2018) Aaron Oord, Sander Dieleman, and Koray Kavukcuoglu Efficient neural audio synthesis. In International Conference on Machine Learning, pages 2410\u20132419"},{"key":"21200_CR27","unstructured":"Pengcheng Guo and Shixing Liu (2022) Chinese speech pretrain. https:\/\/github.com\/TencentGameMate\/chinese_speech_pretrain"},{"key":"21200_CR28","doi-asserted-by":"publisher","unstructured":"Ren Y, Ruan Y, Tan X, Qin T, Zhao S, Zhao Z, and Tie-Yan Liu (2019) Fastspeech: Fast, robust and controllable text to speech. Adv Neural Inf Process Syst, 32 https:\/\/doi.org\/10.48550\/arXiv.1905.09263","DOI":"10.48550\/arXiv.1905.09263"},{"key":"21200_CR29","doi-asserted-by":"crossref","unstructured":"Shen J, Pang R, Weiss RJ, Schuster M, Jaitly N, Yang Z, Chen Z, Zhang Y, Wang Y, Skerry-Ryan RJ, Saurous RA, Agiomyrgiannakis Y, Wu Y (2018) Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions. In 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pages 4779\u20134783","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"21200_CR30","doi-asserted-by":"crossref","unstructured":"Song K, Xue H, Wang X, Cong J, Zhang Y, Xie L, Yang B, Zhang X (2022) and Dan Su Adavits: Tiny vits for low computing resource speaker adaptation. In 2022 13th International Symposium on Chinese Spoken Language Processing (ISCSLP), pages 319\u2013323","DOI":"10.1109\/ISCSLP57327.2022.10037585"},{"key":"21200_CR31","unstructured":"van den Aaron Y, Li I, Babuschkin K, Simonyan O, Vinyals K, van den Kavukcuoglu E, Lockhart LC, Cobo, Demis Hassabis (2018) Florian Stimberg, Norman Casagrande, Dominik Grewe, Seb Noury, Sander Dieleman, Erich Elsen, Nal Kalchbrenner, Heiga Zen, Alex Graves, Helen King, Tom Walters, Dan Belov, and Parallel WaveNet: Fast High-Fidelity Speech Synthesis. PMLR In International conference on machine learning, pages 3918\u20133926"},{"key":"21200_CR32","unstructured":"van den Aaron S, Dieleman H, Zen K, Simonyan O, Vinyals A, Graves Nal Kalchbrenner, Andrew Senior, and Koray Kavukcuoglu (2016) WaveNet: A Generative Model for Raw Audio. arXiv preprint arXiv:1609.03499"},{"key":"21200_CR33","doi-asserted-by":"crossref","unstructured":"Wang P, Qian Y, Frank K, Soong L, He, Zhao H (2015) Word embedding for recurrent neural network based tts synthesis. In 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 4879\u20134883","DOI":"10.1109\/ICASSP.2015.7178898"},{"key":"21200_CR34","doi-asserted-by":"publisher","unstructured":"Wang Y, Skerry-Ryan RJ, Stanton D, Wu Y, Weiss RJ, Jaitly N, Yang Z, Xiao Y, Chen Z, Bengio S, Le Q, Agiomyrgiannakis Y, Clark R, Rif A, Saurous (2017) Tacotron: towards end- to-end speech synthesis. ArXiv Preprint arXiv :170310135 https:\/\/doi.org\/10.48550\/arXiv.1703.10135","DOI":"10.48550\/arXiv.1703.10135"},{"key":"21200_CR35","doi-asserted-by":"publisher","unstructured":"Wei Fang Y-A, Chung, Glass J (2019) Towards transfer learning for end-to-end speech synthesis from deep pre-trained Language models. arXiv preprint arXiv:1906.07307. https:\/\/doi.org\/10.48550\/arXiv.1906.07307","DOI":"10.48550\/arXiv.1906.07307"},{"key":"21200_CR36","unstructured":"Xin Xu Shaoji Zhang Ming Li Yao Shi, and, Bu H (2015) Aishell-3: A multi-speaker mandarin tts corpus and the baselines. https:\/\/arxiv.org\/abs\/2010.11567"},{"key":"21200_CR37","unstructured":"Yi Ren C, Qin HXTT, Zhao S, Zhao Z (2020) and Tie-Yan Liu Fastspeech 2: Fast and high-quality end-to-end text to speech. arXiv preprint arXiv:2006.04558"},{"key":"21200_CR38","doi-asserted-by":"crossref","unstructured":"Yiming Cui W, Che T, Liu B, Qin S, Wang, Guoping H (2020) Revisiting pre-trained models for Chinese natural language processing. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: Findings, pages 657\u2013668. https:\/\/www.aclweb.org\/anthology\/2020.findings-emnlp.58","DOI":"10.18653\/v1\/2020.findings-emnlp.58"},{"key":"21200_CR39","doi-asserted-by":"crossref","unstructured":"Zhan H, Yu X, Zhang H, Zhang Y, Lin Y (2022) Exploring timbre disentanglement in non-autoregressive cross-lingual text-to-speech. In Interspeech","DOI":"10.21437\/Interspeech.2022-205"},{"key":"21200_CR40","doi-asserted-by":"publisher","unstructured":"Zhang B, Lv H, Guo P, Shao Q, Yang C, Xie L, Xu X, Bu H, Chen X, Zeng C, Wu D (2022) Zhendong Peng Wenetspeech: A 10000\u2009+\u2009hours multidomain mandarin corpus for speech recognition. in Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 6182\u20136186 https:\/\/doi.org\/10.48550\/arXiv.2110.03370","DOI":"10.48550\/arXiv.2110.03370"},{"key":"21200_CR41","doi-asserted-by":"crossref","unstructured":"Zhang M, Zhou Y, Zhao L and Haizhou Li(2021)Transfer Learning from Speech Synthesis to Voice Conversion with Non-Parallel Training Data. IEEE\/ACM Transactions on Audio, Speech, and Language Processing. 29, page 1290\u20131302","DOI":"10.1109\/TASLP.2021.3066047"},{"key":"21200_CR42","doi-asserted-by":"crossref","unstructured":"Zhu Y, He J, Jing R, Song Y, Lian J, Zhang Xiao-lei, Li J (2024) LLM-Based Expressive Text-to-Speech Synthesizer with Style and Timbre Disentanglement. in Proceedings of the IEEE\/CVF 14th International Symposium on Chinese Spoken Language Processing (ISCSLP)","DOI":"10.1109\/ISCSLP63861.2024.10800531"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21200-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-026-21200-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21200-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T16:57:36Z","timestamp":1770137856000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-026-21200-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,3]]},"references-count":42,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2026,2]]}},"alternative-id":["21200"],"URL":"https:\/\/doi.org\/10.1007\/s11042-026-21200-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,3]]},"assertion":[{"value":"30 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 November 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 November 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to publish"}},{"value":"The authors have no conflicts of interest to declare that are relevant to the content of this article.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"119"}}