{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T12:11:31Z","timestamp":1779106291210,"version":"3.51.4"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783031162091","type":"print"},{"value":"9783031162107","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-16210-7_24","type":"book-chapter","created":{"date-parts":[[2022,9,20]],"date-time":"2022-09-20T23:03:09Z","timestamp":1663714989000},"page":"294-304","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An End to End Bilingual TTS System for Fongbe and Yoruba"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3078-4286","authenticated-orcid":false,"given":"Charbel Arnaud Cedrique Y.","family":"Boco","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Th\u00e9ophile K.","family":"Dagba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,21]]},"reference":[{"key":"24_CR1","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning WaveNet on mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"24_CR2","unstructured":"Ren, Y., Hu, C., Tan, X., Qin, T., Zhao, S., Zhao, Z., Liu, T.-Y.: Fastspeech 2: fast and high-quality end-to-end text to speech. ArXiv Prepr. ArXiv:200604558 (2020)"},{"key":"24_CR3","unstructured":"Mu, Z., Yang, X., Dong, Y.: Review of end-to-end speech synthesis technology based on deep learning. ArXiv Prepr. ArXiv210409995 (2021)"},{"key":"24_CR4","doi-asserted-by":"crossref","unstructured":"Nekvinda, T., Du\u0161ek, O.: One model, many languages: meta-learning for multilingual text-to-speech. ArXiv Prepr. ArXiv200800768 (2020)","DOI":"10.21437\/Interspeech.2020-2679"},{"key":"24_CR5","unstructured":"Lee, Y., Shon, S., Kim, T.: Learning pronunciation from a foreign language in speech synthesis networks. ArXiv Prepr. ArXiv181109364 (2018)"},{"key":"24_CR6","unstructured":"Cai, Z., Yang, Y., Li, M.: Cross-lingual multispeaker text-to-speech under limited-data scenario. ArXiv Prepr. ArXiv200510441 (2020)"},{"key":"24_CR7","unstructured":"He, M., Yang, J., He, L., Soong, F.K.: Multilingual Byte2Speech models for scalable low-resource speech synthesis. ArXiv Prepr. ArXiv210303541 (2021)"},{"key":"24_CR8","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1016\/j.procs.2014.08.125","volume":"35","author":"TK Dagba","year":"2014","unstructured":"Dagba, T.K., Boco, C.: A text to speech system for fon language using multisyn algorithm. Procedia Comput. Sci. 35, 447\u2013455 (2014)","journal-title":"Procedia Comput. Sci."},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tacotron: towards end-to-end speech synthesis. ArXiv Prepr. ArXiv170310135 (2017)","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"24_CR10","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., Lim, J.: Signal estimation from modified short-time fourier transform. IEEE Trans. Acoust. Speech Signal Process. 32, 236\u2013243 (1984)","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Battenberg, E., Skerry-Ryan, R., Mariooryad, S., Stanton, D., Kao, D., Shannon, M., Bagby, T.: Location-relative attention mechanisms for robust long-form speech synthesis. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6194\u20136198. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054106"},{"key":"24_CR12","unstructured":"Oord, A. van den, et al.: Wavenet: A generative model for raw audio. ArXiv Prepr. ArXiv160903499 (2016)"},{"key":"24_CR13","unstructured":"Ren, Y., et al.: Fastspeech: Fast, robust and controllable text to speech. ArXiv Prepr. ArXiv190509263 (2019)"},{"key":"24_CR14","unstructured":"Kalchbrenner, N., et al.: Efficient neural audio synthesis. In: International Conference on Machine Learning, pp. 2410\u20112419. PMLR (2018)"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"Li, B., Zhang, Y., Sainath, T., Wu, Y., Chan, W.: Bytes are all you need: end-to-end multilingual speech recognition and synthesis with bytes. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5621\u20135625. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682674"},{"key":"24_CR16","doi-asserted-by":"crossref","unstructured":"Cao, Y., et al.: End-to-end code-switched TTS with mix of monolingual recordings. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6935\u20136939. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8682927"},{"key":"24_CR17","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: Learning to speak fluently in a foreign language: multilingual speech synthesis and cross-language voice cloning. ArXiv Prepr. ArXiv190704448 (2019)","DOI":"10.21437\/Interspeech.2019-2668"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Chen, M., et al.: Cross-lingual, multi-speaker text-to-speech synthesis using neural speaker embedding. In: Interspeech, pp. 2105\u20132109 (2019)","DOI":"10.21437\/Interspeech.2019-1632"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Yang, J., He, L.: Towards universal text-to-speech. In: INTERSPEECH, pp. 3171\u20133175 (2020)","DOI":"10.21437\/Interspeech.2020-1590"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Li, N., Liu, S., Liu, Y., Zhao, S., Liu, M.: Neural speech synthesis with transformer network. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 6706\u20136713 (2019)","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"24_CR21","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1007\/s10772-016-9334-8","volume":"19","author":"JOR Aoga","year":"2016","unstructured":"Aoga, J.O.R., Dagba, T.K., Fanou, C.C.: Integration of Yoruba language into MaryTTS. Int. J. Speech Technol. 19, 151\u2013158 (2016)","journal-title":"Int. J. Speech Technol."},{"issue":"36","key":"24_CR22","doi-asserted-by":"publisher","first-page":"13","DOI":"10.5120\/cae2021652884","volume":"7","author":"AE Akinwonm","year":"2021","unstructured":"Akinwonm, A.E.: Development of a prosodic read speech syllabic corpus of the Yoruba language. Commun. Appl. Electron. CAE Found. Comput. Sci. FCS 7(36), 13\u201332 (2021). https:\/\/doi.org\/10.5120\/cae2021652884","journal-title":"Commun. Appl. Electron. CAE Found. Comput. Sci. FCS"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Gutkin, A., Demir\u015fahin, I., Kjartansson, O., Rivera, C., T\u00fabd\u00f2s\u00fan, K.: Developing an open-source corpus of Yoruba speech. In: Proceedings of Interspeech 2020, pp. 404\u2013408. International Speech and Communication Association (ISCA), Shanghai, China (2020)","DOI":"10.21437\/Interspeech.2020-1096"},{"key":"24_CR24","unstructured":"Kumar, K., et al.: Melgan: Generative adversarial networks for conditional waveform synthesis. ArXiv Prepr. ArXiv191006711 (2019)"},{"key":"24_CR25","unstructured":"TensorFlowTTS: https:\/\/github.com\/tensorspeech\/TensorFlowTTS. Last accessed 1 June 2022"},{"key":"24_CR26","unstructured":"Eberhard, D.M., Simons, G.F., Fennig, C.D.: Ethnologue: Languages of the World. Twenty-fifth edition. Dallas, Texas: SIL International. Online version: http:\/\/www.ethnologue.com (2022). Last accessed 1 June 2022"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Laleye, F.A., Besacier, L., Ezin, E.C., Motamed, C.: First automatic fongbe continuous speech recognition system: development of acoustic models and language models. In: 2016 Federated Conference on Computer Science and Information Systems (FedCSIS), pp. 477\u2013482. IEEE (2016)","DOI":"10.15439\/2016F153"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M.: Montreal forced aligner: trainable text-speech alignment using Kaldi. In: Interspeech, pp. 498\u2013502 (2017)","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"24_CR29","unstructured":"Ito, K., Johnson, L.: The LJ Speech Dataset. 2017. https:\/\/keithito.com\/LJ-Speech-Dataset (2017). Last accessed 1 June 2022"},{"key":"24_CR30","doi-asserted-by":"crossref","unstructured":"Kubichek, R.: Mel-cepstral distance measure for objective speech quality assessment. In: Proceedings of IEEE Pacific Rim Conference on Communications Computers and Signal Processing, pp. 125\u2013128. IEEE (1993)","DOI":"10.1109\/PACRIM.1993.407206"}],"container-title":["Communications in Computer and Information Science","Advances in Computational Collective Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-16210-7_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,4]],"date-time":"2024-10-04T06:08:12Z","timestamp":1728022092000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-16210-7_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031162091","9783031162107"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-16210-7_24","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"21 September 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}