{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T16:27:00Z","timestamp":1779208020632,"version":"3.51.4"},"publisher-location":"Cham","reference-count":39,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031404979","type":"print"},{"value":"9783031404986","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-40498-6_17","type":"book-chapter","created":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T23:02:34Z","timestamp":1692745354000},"page":"188-199","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["CML-TTS: A\u00a0Multilingual Dataset for\u00a0Speech Synthesis in\u00a0Low-Resource Languages"],"prefix":"10.1007","author":[{"given":"Frederico S.","family":"Oliveira","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edresson","family":"Casanova","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arnaldo Candido","family":"Junior","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anderson S.","family":"Soares","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arlindo R.","family":"Galv\u00e3o Filho","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,23]]},"reference":[{"key":"17_CR1","unstructured":"Ardila, R., et al.: Common voice: a massively-multilingual speech corpus. In: Proceedings of the Twelfth Language Resources and Evaluation Conference, pp. 4218\u20134222. European Language Resources Association, Marseille, France, May 2020. https:\/\/aclanthology.org\/2020.lrec-1.520"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Casanova, E., et al.: TTS-Portuguese Corpus: a corpus for speech synthesis in Brazilian Portuguese. Lang. Resour. Eval. 1\u201313 (2022)","DOI":"10.1007\/s10579-021-09570-4"},{"key":"17_CR3","doi-asserted-by":"publisher","unstructured":"Casanova, E., et al.: SC-GlowTTS: an efficient zero-shot multi-speaker text-to-speech model (2021). https:\/\/doi.org\/10.48550\/ARXIV.2104.05557, https:\/\/arxiv.org\/abs\/2104.05557","DOI":"10.48550\/ARXIV.2104.05557"},{"key":"17_CR4","unstructured":"Casanova, E., Weber, J., Shulby, C.D., Junior, A.C., G\u00f6lge, E., Ponti, M.A.: YourTTS: towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone. In: International Conference on Machine Learning, pp. 2709\u20132720. PMLR (2022)"},{"key":"17_CR5","unstructured":"Chung, J.S., et al.: In defence of metric learning for speaker recognition. arXiv preprint arXiv:2003.11982 (2020)"},{"key":"17_CR6","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Nagrani, A., Zisserman, A.: VoxCeleb2: deep speaker recognition. CoRR abs\/1806.05622 (2018). http:\/\/arxiv.org\/abs\/1806.05622","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"17_CR7","doi-asserted-by":"publisher","unstructured":"Conneau, A., Baevski, A., Collobert, R., Mohamed, A., Auli, M.: Unsupervised cross-lingual representation learning for speech recognition. In: Hermansky, H., Cernock\u00fd, H., Burget, L., Lamel, L., Scharenborg, O., Motl\u00edcek, P. (eds.) Interspeech 2021, 22nd Annual Conference of the International Speech Communication Association, Brno, Czechia, 30 August\u20133 September 2021, pp. 2426\u20132430. ISCA (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-329","DOI":"10.21437\/Interspeech.2021-329"},{"issue":"3","key":"17_CR8","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1049\/et.2017.0330","volume":"12","author":"P Dempsey","year":"2017","unstructured":"Dempsey, P.: The teardown: google home personal assistant. Eng. Technol. 12(3), 80\u201381 (2017)","journal-title":"Eng. Technol."},{"key":"17_CR9","unstructured":"Goodfellow, I., Bengio, Y., Courville, A., Bengio, Y.: Deep Learning, vol. 1. MIT press, Cambridge (2016)"},{"key":"17_CR10","unstructured":"Gruber, T.R.: Siri, a virtual personal assistant-bringing intelligence to the interface. In: Semantic Technologies Conference (2009)"},{"key":"17_CR11","doi-asserted-by":"publisher","unstructured":"Heo, H.S., Lee, B.J., Huh, J., Chung, J.S.: Clova Baseline System for the VoxCeleb Speaker Recognition Challenge 2020 (2020). https:\/\/doi.org\/10.48550\/ARXIV.2009.14153, https:\/\/arxiv.org\/abs\/2009.14153","DOI":"10.48550\/ARXIV.2009.14153"},{"key":"17_CR12","doi-asserted-by":"crossref","unstructured":"Huang, R., Zhao, Z., Liu, H., Liu, J., Cui, C., Ren, Y.: ProDiff: progressive fast diffusion model for high-quality text-to-speech. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 2595\u20132605 (2022)","DOI":"10.1145\/3503161.3547855"},{"key":"17_CR13","unstructured":"Ito, K., Johnson, L.: The LJSpeech Dataset (2017). https:\/\/keithito.com\/LJ-Speech-Dataset\/"},{"key":"17_CR14","unstructured":"Jemine, C.: Master Thesis: Real-time voice cloning. Master\u2019s thesis, Facult\u00e9 des Sciences Appliqu\u00e8es (2019)"},{"key":"17_CR15","doi-asserted-by":"crossref","unstructured":"Kim, C., Stern, R.M.: Robust signal-to-noise ratio estimation based on waveform amplitude distribution analysis. In: Ninth Annual Conference of the International Speech Communication Association (2008)","DOI":"10.21437\/Interspeech.2008-644"},{"key":"17_CR16","first-page":"8067","volume":"33","author":"J Kim","year":"2020","unstructured":"Kim, J., Kim, S., Kong, J., Yoon, S.: Glow-TTS: a generative flow for text-to-speech via monotonic alignment search. Adv. Neural Inf. Process. Syst. 33, 8067\u20138077 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"17_CR17","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In: International Conference on Machine Learning, pp. 5530\u20135540. PMLR (2021)"},{"key":"17_CR18","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high fidelity speech synthesis (2020)"},{"key":"17_CR19","doi-asserted-by":"crossref","unstructured":"Li, B., Zhang, Y., Sainath, T., Wu, Y., Chan, W.: Bytes are all you need: end-to-end multilingual speech recognition and synthesis with bytes. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5621\u20135625. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682674"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Lux, F., Vu, N.T.: Language-agnostic meta-learning for low-resource text-to-speech with articulatory features. arXiv preprint arXiv:2203.03191 (2022)","DOI":"10.18653\/v1\/2022.acl-long.472"},{"key":"17_CR21","unstructured":"Munich Artificial Intelligence Laboratories GmbH: The M-AILABS Speech Dataset (2017). https:\/\/www.caito.de\/2019\/01\/03\/the-m-ailabs-speech-dataset\/feld.de\/content\/bworld-robot-control-software\/. Accessed 05 Nov 2022"},{"key":"17_CR22","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: Voxceleb: a large-scale speaker identification dataset. In: INTERSPEECH (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"17_CR23","doi-asserted-by":"crossref","unstructured":"Nekvinda, T., Du\u0161ek, O.: One model, many languages: meta-learning for multilingual text-to-speech. arXiv preprint arXiv:2008.00768 (2020)","DOI":"10.21437\/Interspeech.2020-2679"},{"key":"17_CR24","unstructured":"van den Oord, A., et al: WaveNet: a generative model for raw audio. CoRR abs\/1609.03499 (2016). http:\/\/arxiv.org\/abs\/1609.03499"},{"key":"17_CR25","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: an ASR corpus based on public domain audio books. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5206\u20135210 (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"17_CR26","unstructured":"Ping, W., et al.: Deep voice 3: 2000-speaker neural text-to-speech. arXiv preprint arXiv:1710.07654 (2017)"},{"key":"17_CR27","doi-asserted-by":"publisher","unstructured":"Pratap, V., Xu, Q., Sriram, A., Synnaeve, G., Collobert, R.: MLS: a large-scale multilingual dataset for speech research. In: Interspeech 2020. ISCA, October 2020. https:\/\/doi.org\/10.21437\/interspeech.2020-2826, https:\/\/doi.org\/10.21437%2Finterspeech.2020-2826","DOI":"10.21437\/interspeech.2020-2826"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Purington, A., Taft, J.G., Sannon, S., Bazarova, N.N., Taylor, S.H.: Alexa is my new BFF social roles, user satisfaction, and personification of the Amazon Echo. In: Proceedings of the 2017 CHI Conference Extended Abstracts on Human Factors in Computing Systems, pp. 2853\u20132859 (2017)","DOI":"10.1145\/3027063.3053246"},{"key":"17_CR29","doi-asserted-by":"crossref","unstructured":"Salesky, E., et al.: The multilingual TEDx corpus for speech recognition and translation. CoRR abs\/2102.01757 (2021). https:\/\/arxiv.org\/abs\/2102.01757","DOI":"10.21437\/Interspeech.2021-11"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning WaveNet on Mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"17_CR31","unstructured":"Sotelo, J., et al.: Char2Wav: end-to-end speech synthesis. In: International Conference on Learning Representations, Workshop (2017)"},{"key":"17_CR32","doi-asserted-by":"crossref","unstructured":"Tachibana, H., Uenoyama, K., Aihara, S.: Efficiently trainable text-to-speech system based on deep convolutional networks with guided attention. arXiv preprint arXiv:1710.08969 (2017)","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"17_CR33","unstructured":"Valle, R., Shih, K.J., Prenger, R., Catanzaro, B.: Flowtron: an autoregressive flow-based generative network for text-to-speech synthesis. In: International Conference on Learning Representations (2020)"},{"key":"17_CR34","doi-asserted-by":"crossref","unstructured":"Wan, L., Wang, Q., Papir, A., Moreno, I.L.: Generalized end-to-end loss for speaker verification. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4879\u20134883. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"17_CR35","unstructured":"Wang, Y., et al.: Tacotron: a fully end-to-end text-to-speech synthesis model. arXiv preprint arXiv:1703.10135 (2017)"},{"key":"17_CR36","unstructured":"Yamagishi, J., Veaux, C., MacDonald, K.: CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit (2019). https:\/\/datashare.ed.ac.uk\/handle\/10283\/3443. Accessed 05 Nov 2022"},{"key":"17_CR37","doi-asserted-by":"crossref","unstructured":"Ze, H., Senior, A., Schuster, M.: Statistical parametric speech synthesis using deep neural networks. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 7962\u20137966. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"17_CR38","doi-asserted-by":"crossref","unstructured":"Zen, H., et al.: LibriTTS: a corpus derived from LibriSpeech for text-to-speech. arXiv preprint arXiv:1904.02882 (2019)","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"17_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: Learning to speak fluently in a foreign language: multilingual speech synthesis and cross-language voice cloning. arXiv preprint arXiv:1907.04448 (2019)","DOI":"10.21437\/Interspeech.2019-2668"}],"container-title":["Lecture Notes in Computer Science","Text, Speech, and Dialogue"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-40498-6_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T16:19:38Z","timestamp":1729959578000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-40498-6_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031404979","9783031404986"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-40498-6_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"23 August 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"TSD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Text, Speech, and Dialogue","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pilsen","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Czech Republic","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"tsd2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.kiv.zcu.cz\/tsd2023\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMS & back-office system","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"64","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"31","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.56","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}