{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:16:50Z","timestamp":1783192610289,"version":"3.54.6"},"publisher-location":"Singapore","reference-count":11,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819549627","type":"print"},{"value":"9789819549634","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T00:00:00Z","timestamp":1763769600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T00:00:00Z","timestamp":1763769600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-4963-4_18","type":"book-chapter","created":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T17:07:09Z","timestamp":1763744829000},"page":"213-225","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Cost-Effective Voice Cloning System for\u00a0Vietnamese TTS: A Case Study at\u00a0HCMUT"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0014-6317","authenticated-orcid":false,"given":"Vinh Q.","family":"Vo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bao G.","family":"Quach","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quyen T.","family":"Bui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Khai Q.","family":"Truong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7488-4714","authenticated-orcid":false,"given":"Long S. T.","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fabien","family":"Baldacci","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0467-6254","authenticated-orcid":false,"given":"Tho T.","family":"Quan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,22]]},"reference":[{"key":"18_CR1","unstructured":"Casanova, E., Weber, J., Shulby, C.D., Junior, A.C., G\u00f6lge, E., Ponti, M.A.: YourTTS: towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone. In: Chaudhuri, K., Jegelka, S., Song, L., Szepesvari, C., Niu, G., Sabato, S. (eds.) Proceedings of the 39th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0162, pp. 2709\u20132720. PMLR (2022). https:\/\/proceedings.mlr.press\/v162\/casanova22a.html"},{"key":"18_CR2","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"WN Hsu","year":"2021","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: HuBERT: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3451\u20133460 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3122291","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"18_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.wocn.2018.07.001","volume":"71","author":"Y Jadoul","year":"2018","unstructured":"Jadoul, Y., Thompson, B., de Boer, B.: Introducing parselmouth: a python interface to praat. J. Phon. 71, 1\u201315 (2018). https:\/\/doi.org\/10.1016\/j.wocn.2018.07.001","journal-title":"J. Phon."},{"key":"18_CR4","unstructured":"Jia, Y., et al.: Transfer learning from speaker verification to multispeaker text-to-speech synthesis. In: Bengio, S., Wallach, H., Larochelle, H., Grauman, K., Cesa-Bianchi, N., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol.\u00a031. Curran Associates, Inc. (2018). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2018\/file\/6832a7b24bc06775d02b7406880b93fc-Paper.pdf"},{"key":"18_CR5","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 5530\u20135540. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/kim21f.html"},{"key":"18_CR6","unstructured":"van\u00a0den Oord, A., et al.: WaveNet: a generative model for raw audio (2016). https:\/\/arxiv.org\/abs\/1609.03499"},{"key":"18_CR7","unstructured":"Ren, Y., et al.: Fastspeech 2: fast and high-quality end-to-end text to speech (2020). https:\/\/arxiv.org\/abs\/2006.04558. Accepted by ICLR 2021; Latest version: v8, August 2022"},{"key":"18_CR8","doi-asserted-by":"publisher","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning wavenet on mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783 (2018). https:\/\/doi.org\/10.1109\/ICASSP.2018.8461368","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"18_CR9","unstructured":"svc-develop-team: SoftVC VITS Singing Voice Conversion. https:\/\/github.com\/svc-develop-team\/so-vits-svc (2023). Archived repository. Accessed 04 Oct 2025"},{"key":"18_CR10","doi-asserted-by":"publisher","unstructured":"Tokuda, K., Yoshimura, T., Masuko, T., Kobayashi, T., Kitamura, T.: Speech parameter generation algorithms for HMM-based speech synthesis. In: IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 3, pp. 1315\u20131318 (2000). https:\/\/doi.org\/10.1109\/ICASSP.2000.861820","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"18_CR11","doi-asserted-by":"publisher","unstructured":"\u0141a\u0144cucki, A.: Fastpitch: parallel text-to-speech with pitch prediction. In: ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6588\u20136592 (2021). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9413889","DOI":"10.1109\/ICASSP39728.2021.9413889"}],"container-title":["Lecture Notes in Computer Science","Multi-disciplinary Trends in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-4963-4_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T18:58:36Z","timestamp":1783191516000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-4963-4_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,22]]},"ISBN":["9789819549627","9789819549634"],"references-count":11,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-4963-4_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,22]]},"assertion":[{"value":"22 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"This study complies with institutional and academic ethical standards governing data collection and dissemination. Written informed consent was obtained from all participants, and all personally identifying metadata was removed. Publicly available materials are limited to short audio excerpts, while full-length recordings are stored under restricted access and may be shared only upon justified request. To reduce risks of misuse, all released synthetic audio is watermarked, model checkpoints are distributed solely for approved research purposes, and the project repository includes a responsible-use statement delineating acceptable applications and explicitly prohibiting malicious use.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics and Consent"}},{"value":"MIWAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multi-disciplinary Trends in Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ho Chi Minh City","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vietnam","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 December 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miwai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/miwai25.miwai.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}