{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T21:16:28Z","timestamp":1757625388791,"version":"3.44.0"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783032025470"},{"type":"electronic","value":"9783032025487"}],"license":[{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-02548-7_18","type":"book-chapter","created":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T05:39:42Z","timestamp":1755754782000},"page":"207-217","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["How Far Can Synthetic Speech Go? Enhancing ASR in\u00a0Low-Resource Scenarios via\u00a0Voice Cloning"],"prefix":"10.1007","author":[{"given":"Dalai","family":"Mengke","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Meng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"P\u00e9ter","family":"Mihajlik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,22]]},"reference":[{"key":"18_CR1","doi-asserted-by":"crossref","unstructured":"Ahlawat, H., Aggarwal, N., Gupta, D.: Automatic speech recognition: a survey of deep learning techniques and approaches. Int. J. Cogn. Comput. Eng. (2025)","DOI":"10.1016\/j.ijcce.2024.12.007"},{"key":"18_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2023.101538","volume":"83","author":"D O\u2019Shaughnessy","year":"2024","unstructured":"O\u2019Shaughnessy, D.: Trends and developments in automatic speech recognition research. Comput. Speech Lang. 83, 101538 (2024)","journal-title":"Comput. Speech Lang."},{"key":"18_CR3","unstructured":"Yeroyan, A., Karpov, N.: Enabling ASR for low-resource languages: a comprehensive dataset creation approach. arXiv preprint arXiv:2406.01446 (2024)"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Reitmaier, T., et al.: Opportunities and challenges of automatic speech recognition systems for low-resource language speakers. In: Proceedings of the 2022 CHI Conference on Human Factors in Computing Systems, pp. 1\u201317 (2022)","DOI":"10.1145\/3491102.3517639"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Singh, V.P., Sailor, H., Bhattacharya, S., Pandey, A.: Spectral modification based data augmentation for improving end-to-end ASR for children\u2019s speech. arXiv preprint arXiv:2203.06600 (2022)","DOI":"10.21437\/Interspeech.2022-11343"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Ko, T., Peddinti, V., Povey, D., Khudanpur, S.: Audio augmentation for speech recognition. In: Interspeech, vol. 2015, p. 3586 (2015)","DOI":"10.21437\/Interspeech.2015-711"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Park, D.S., et al.: Specaugment: a simple data augmentation method for automatic speech recognition. arXiv preprint arXiv:1904.08779 (2019)","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Ko, T., Peddinti, V., Povey, D., Seltzer, M.L., Khudanpur, S.: A study on data augmentation of reverberant speech for robust speech recognition. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5220\u20135224. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Chao, F.A., Jiang, S. W.F., Yan, B.C., Hung, J. W., Chen, B.: Tenet: a time-reversal enhancement network for noise-robust ASR. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 55\u201361. IEEE (2021)","DOI":"10.1109\/ASRU51503.2021.9687924"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Bartelds, M., San, N., McDonnell, B., Jurafsky, D., Wieling, M.: Making more of little data: improving low-resource automatic speech recognition using data augmentation. arXiv preprint arXiv:2305.10951 (2023)","DOI":"10.18653\/v1\/2023.acl-long.42"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Ratnarajah, A., Tang, Z., Manocha, D.: TS-RIR: translated synthetic room impulse responses for speech augmentation. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 259\u2013266. IEEE (2021)","DOI":"10.1109\/ASRU51503.2021.9688304"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Singh, D.K., Amin, P.P., Sailor, H.B., Patil, H.A.: Data augmentation using cyclegan for end-to-end children ASR. In: 2021 29th European Signal Processing Conference (EUSIPCO), pp. 511\u2013515. IEEE (2021)","DOI":"10.23919\/EUSIPCO54536.2021.9616228"},{"key":"18_CR13","doi-asserted-by":"crossref","unstructured":"Mimura, M., Ueno, S., Inaguma, H., Sakai, S., Kawahara, T.: Leveraging sequence-to-sequence speech synthesis for enhancing acoustic-to-word speech recognition. In: 2018 IEEE Spoken Language Technology Workshop (SLT), pp. 477\u2013484. IEEE (2018)","DOI":"10.1109\/SLT.2018.8639589"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Ueno, S., Mimura, M., Sakai, S., Kawahara, T.: Multi-speaker sequence-to-sequence speech synthesis for data augmentation in acoustic-to-word speech recognition. In: ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6161\u20136165. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682816"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Zevallos, R., Bel, N., C\u00e1mbara, G., Farr\u00fas, M., Luque, J.: Data augmentation for low-resource Quechua ASR improvement. arXiv preprint arXiv:2207.06872 (2022)","DOI":"10.21437\/Interspeech.2022-770"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Laptev, A., Korostik, R., Svischev, A., Andrusenko, A., Medennikov, I., Rybin, S.: You do not need more data: improving end-to-end speech recognition by text-to-speech data augmentation. In: 2020 13th International Congress on Image and Signal Processing, BioMedical Engineering and Informatics (CISP-BMEI), pp. 439\u2013444. IEEE (2020)","DOI":"10.1109\/CISP-BMEI51763.2020.9263564"},{"key":"18_CR17","unstructured":"Du, Z., et\u00a0al. Cosyvoice: a scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens. arXiv preprint arXiv:2407.05407 (2024)"},{"key":"18_CR18","unstructured":"Casanova, E., Weber, J., Shulby, C.D., Junior, A.C., G\u00f6lge, E., Ponti, M.A.: Yourtts: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone. In: International Conference on Machine Learning, pp. 2709\u20132720. PMLR (2022)"},{"key":"18_CR19","unstructured":"Chen, S., et al.: Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers. arXiv preprint arXiv:2406.05370 (2024)"},{"key":"18_CR20","unstructured":"Le, M., et al.: Voicebox: text-guided multilingual universal speech generation at scale. In: Advances in Neural Information Processing Systems, vol. 36, pp. 14005\u201314034 (2023)"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Casanova, E., et\u00a0al. Xtts: a massively multilingual zero-shot text-to-speech model. arXiv preprint arXiv:2406.04904 (2024)","DOI":"10.21437\/Interspeech.2024-2016"},{"key":"18_CR22","doi-asserted-by":"crossref","unstructured":"Casanova, E., et al.: ASR data augmentation in low-resource settings using cross-lingual multi-speaker TTS and cross-lingual voice conversion. arXiv preprint arXiv:2204.00618 (2022)","DOI":"10.21437\/Interspeech.2023-496"},{"key":"18_CR23","doi-asserted-by":"crossref","unstructured":"Yang, G., et al.: Enhancing low-resource ASR through versatile TTS: bridging the data gap. In: ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2025)","DOI":"10.1109\/ICASSP49660.2025.10889894"},{"key":"18_CR24","unstructured":"Azzuni, H., El Saddik, A.: Voice cloning: comprehensive survey. arXiv preprint arXiv:2505.00579 (2025)"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"Li, R., Pu, D., Huang, M., Huang, B.: Unet-TTS: improving unseen speaker and style transfer in one-shot voice cloning. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8327\u20138331. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746049"},{"key":"18_CR26","unstructured":"Neekhara, P., Hussain, S., Dubnov, S., Koushanfar, F., McAuley, J.: Expressive neural voice cloning. In: Asian Conference on Machine Learning, pp. 252\u2013267. PMLR (2021)"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Park, K., Mulc, T.: Css10: a collection of single speaker speech datasets for 10 languages. arXiv preprint arXiv:1903.11269 (2019)","DOI":"10.21437\/Interspeech.2019-1500"},{"key":"18_CR28","unstructured":"Oravecz, C., V\u00e1radi, T., Sass, B.: The hungarian gigaword corpus (2014)"},{"key":"18_CR29","unstructured":"Mihajlik, P., Balog, A., Gr\u00e1czi, T.E., Koh\u00e1ri, A., Tarj\u00e1n, B., M\u00e1dy, K.: Bea-base: a benchmark for ASR of spontaneous Hungarian. arXiv preprint arXiv:2202.00601 (2022)"},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Gulati, A., et\u00a0al.: Conformer: convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"18_CR31","unstructured":"Kuchaiev, O., et\u00a0al. Nemo: a toolkit for building AI applications using neural modules. arXiv preprint arXiv:1909.09577 (2019)"},{"key":"18_CR32","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"18_CR33","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning, pp. 369\u2013376 (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"18_CR34","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., Birch, A.: Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909 (2015)","DOI":"10.18653\/v1\/P16-1162"},{"key":"18_CR35","doi-asserted-by":"crossref","unstructured":"Mihajlik, P., et al.: What kind of multi-or cross-lingual pre-training is the most effective for a spontaneous, less-resourced ASR task? In: 2nd Annual Meeting of the ELRA\/ISCA Special Interest Group on Under-resourced Languages (SIGUL 2023) (2023)","DOI":"10.21437\/SIGUL.2023-13"}],"container-title":["Lecture Notes in Computer Science","Text, Speech, and Dialogue"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-02548-7_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T18:05:30Z","timestamp":1757441130000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-02548-7_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,22]]},"ISBN":["9783032025470","9783032025487"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-02548-7_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025,8,22]]},"assertion":[{"value":"22 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"TSD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Text, Speech, and Dialogue","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Erlangen","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"tsd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.kiv.zcu.cz\/tsd2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}