{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T12:56:58Z","timestamp":1742993818903,"version":"3.40.3"},"publisher-location":"Cham","reference-count":31,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030661502"},{"type":"electronic","value":"9783030661519"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-66151-9_9","type":"book-chapter","created":{"date-parts":[[2020,12,21]],"date-time":"2020-12-21T00:02:47Z","timestamp":1608508967000},"page":"141-153","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Text-to-Speech Duration Models for Resource-Scarce Languages in Neural Architectures"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8168-7857","authenticated-orcid":false,"given":"Johannes A.","family":"Louw","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,12,21]]},"reference":[{"key":"9_CR1","doi-asserted-by":"crossref","unstructured":"Black, A.W., Tokuda, K.: The blizzard challenge-2005: evaluating corpus-based speech synthesis on common datasets. In: 9th European Conference on Speech Communication and Technology, pp. 77\u201380 (September 2005)","DOI":"10.21437\/Interspeech.2005-72"},{"key":"9_CR2","unstructured":"Campbell, W.N.: Syllable-based segmental duration. In: Talking Machines: Theories, Models, and Designs, pp. 211\u2013224 (1992)"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Delving deep into rectifiers: surpassing human-level performance on ImageNet classification. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (December 2015)","DOI":"10.1109\/ICCV.2015.123"},{"key":"9_CR4","doi-asserted-by":"crossref","unstructured":"Jiang, Y., et al.: The USTC system for blizzard challenge 2019. In: Blizzard Challenge Workshop 2019, Vienna, Austria (September 2019)","DOI":"10.21437\/Blizzard.2019-20"},{"key":"9_CR5","unstructured":"Kalchbrenner, N., et al.: Efficient Neural Audio Synthesis. arXiv e-prints arXiv:1802.08435 (February 2018)"},{"key":"9_CR6","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"issue":"4","key":"9_CR7","doi-asserted-by":"publisher","first-page":"1102","DOI":"10.1121\/1.1914322","volume":"54","author":"DH Klatt","year":"1973","unstructured":"Klatt, D.H.: Interaction between two factors that influence vowel duration. J. Acoust. Soc. Am. 54(4), 1102\u20131104 (1973)","journal-title":"J. Acoust. Soc. Am."},{"key":"9_CR8","unstructured":"Louw, J.A.: Neural speech synthesis for resource-scarce languages. In: Barnard, E., Davel, M. (eds.) Proceedings of the South African Forum for Artificial Intelligence Research, Cape Town, South Africa, pp. 103\u2013116 (December 2019)"},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Louw, J.A., Moodley, A., Govender, A.: The speect text-to-speech entry for the blizzard challenge 2016. In: Blizzard Challenge Workshop 2016, Cupertino, United States of America (September 2016)","DOI":"10.21437\/Blizzard.2016-9"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Louw, J.A., van Niekerk, D.R., Schl\u00fcnz, G.: Introducing the speect speech synthesis platform. In: Blizzard Challenge Workshop 2010, Kyoto, Japan (September 2010)","DOI":"10.21437\/Blizzard.2010-4"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Morais, E., Violaro, F.: Exploratory analysis of linguistic data based on genetic algorithm for robust modeling of the segmental duration of speech. In: 9th European Conference on Speech Communication and Technology (2005)","DOI":"10.14209\/sbrt.2005.434"},{"key":"9_CR12","unstructured":"van Niekerk, D., de Waal, A., Schl\u00fcnz, G.: Lwazi II Afrikaans TTS Corpus (November 2015). https:\/\/repo.sadilar.org\/handle\/20.500.12185\/443. ISLRN: 570\u2013884-577-153-6"},{"key":"9_CR13","unstructured":"Ren, Y., Hu, C., Qin, T., Zhao, S., Zhao, Z., Liu, T.Y.: FastSpeech 2: Fast and High-Quality End-to-End Text-to-Speech. arXiv preprint arXiv:2006.04558 (2020)"},{"key":"9_CR14","unstructured":"Ren, Y., et al.: Fastspeech: fast, robust and controllable text to speech. In: Advances in Neural Information Processing Systems, pp. 3171\u20133180 (2019)"},{"key":"9_CR15","unstructured":"Riley, M.D.: Tree-based modelling for speech synthesis. In: The ESCA Workshop on Speech Synthesis, pp. 229\u2013232 (1991)"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions. arXiv e-prints arXiv:1712.05884 (December 2017)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"9_CR17","doi-asserted-by":"crossref","unstructured":"Silverman, K., et al.: ToBI: a standard for labeling English prosody. In: Proceedings of the 2nd International Conference on Spoken Language Processing (ICSLP), Alberta, Canada, pp. 867\u2013870 (October 1992)","DOI":"10.21437\/ICSLP.1992-260"},{"key":"9_CR18","unstructured":"Sotelo, J., et al.: Char2wav: End-to-end speech synthesis. arXiv preprint arXiv:1609.03499 (2017)"},{"key":"9_CR19","doi-asserted-by":"crossref","unstructured":"Tachibana, H., Uenoyama, K., Aihara, S.: Efficiently trainable text-to-speech system based on deep convolutional networks with guided attention. arXiv e-prints arXiv:1710.08969 (October 2017)","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"9_CR20","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338","volume-title":"Text-to-Speech Synthesis","author":"P Taylor","year":"2009","unstructured":"Taylor, P.: Text-to-Speech Synthesis. Cambridge University Press, Cambridge (2009)"},{"issue":"5","key":"9_CR21","doi-asserted-by":"publisher","first-page":"1234","DOI":"10.1109\/JPROC.2013.2251852","volume":"101","author":"K Tokuda","year":"2013","unstructured":"Tokuda, K., Nankaku, Y., Toda, T., Zen, H., Yamagishi, J., Oura, K.: Speech synthesis based on hidden Markov models. Proc. IEEE 101(5), 1234\u20131252 (2013)","journal-title":"Proc. IEEE"},{"key":"9_CR22","unstructured":"van den Oord, A., et al.: WaveNet: A generative model for raw audio. arXiv e-prints arXiv:1609.03499 (September 2016)"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tacotron: Towards end-to-end speech synthesis. arXiv e-prints arXiv:1703.10135 (March 2017)","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Watts, O., Henter, G.E., Fong, J., Valentini-Botinhao, C.: Where do the improvements come from in sequence-to-sequence neural TTS? In: 10th ISCA Speech Synthesis Workshop, ISCA, Vienna, Austria (September 2019)","DOI":"10.21437\/SSW.2019-39"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Wei, X., Hunt, M., Skilling, A.: Neural network-based modeling of phonetic durations. arXiv preprint arXiv:1909.03030 (2019)","DOI":"10.21437\/Interspeech.2019-2102"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Wu, Z., Watts, O., King, S.: Merlin: an open source neural network speech synthesis system. In: SSW, pp. 202\u2013207 (2016)","DOI":"10.21437\/SSW.2016-33"},{"key":"9_CR27","first-page":"175","volume-title":"The HTK Book","author":"S Young","year":"2002","unstructured":"Young, S., et al.: The HTK Book, vol. 3, p. 175. Cambridge University Engineering Department, Cambridge (2002)"},{"issue":"5","key":"9_CR28","doi-asserted-by":"publisher","first-page":"825","DOI":"10.1093\/ietisy\/e90-d.5.825","volume":"E90\u2013D","author":"H Zen","year":"2007","unstructured":"Zen, H., Tokuda, K., Masuko, T., Kobayasih, T., Kitamura, T.: A hidden semi-Markov model-based speech synthesis system. IEICE Trans. Inf. Syst. E90\u2013D(5), 825\u2013834 (2007)","journal-title":"IEICE Trans. Inf. Syst."},{"key":"9_CR29","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A.: Deep mixture density networks for acoustic modeling in statistical parametric speech synthesis. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 3844\u20133848. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6854321"},{"issue":"5","key":"9_CR30","doi-asserted-by":"publisher","first-page":"825","DOI":"10.1093\/ietisy\/e90-d.5.825","volume":"E90\u2013D","author":"H Zen","year":"2007","unstructured":"Zen, H., Tokuda, K., Masuko, T., Kobayasih, T., Kitamura, T.: A hidden semi-Markov model-based speech synthesis system. IEICE Trans. Infor. Sys. E90\u2013D(5), 825\u2013834 (2007)","journal-title":"IEICE Trans. Infor. Sys."},{"key":"9_CR31","doi-asserted-by":"publisher","first-page":"65955","DOI":"10.1109\/ACCESS.2019.2914149","volume":"7","author":"X Zhu","year":"2019","unstructured":"Zhu, X., Zhang, Y., Yang, S., Xue, L., Xie, L.: Pre-alignment guided attention for improving training efficiency and model stability in end-to-end speech synthesis. IEEE Access 7, 65955\u201365964 (2019)","journal-title":"IEEE Access"}],"container-title":["Communications in Computer and Information Science","Artificial Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-66151-9_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,19]],"date-time":"2024-08-19T17:39:52Z","timestamp":1724089192000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-66151-9_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030661502","9783030661519"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-66151-9_9","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"21 December 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SACAIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Southern African Conference for Artificial Intelligence Research","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Muldersdrift","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"South Africa","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 February 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 February 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"sacair2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/sacair.org.za\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"53","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"19","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"36% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Dur to the COVID-19 pandemic SACAIR 2020 was postponed to February 2021","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}