{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,22]],"date-time":"2025-03-22T11:45:04Z","timestamp":1742643904079,"version":"3.37.3"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T00:00:00Z","timestamp":1641945600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T00:00:00Z","timestamp":1641945600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100002322","name":"Coordination for the Improvement of Higher Education Personnel","doi-asserted-by":"crossref","award":["88887.464567\/2019-00"],"award-info":[{"award-number":["88887.464567\/2019-00"]}],"id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["304266\/2020-5"],"award-info":[{"award-number":["304266\/2020-5"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Lang Resources &amp; Evaluation"],"published-print":{"date-parts":[[2022,9]]},"DOI":"10.1007\/s10579-021-09570-4","type":"journal-article","created":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T00:02:41Z","timestamp":1641945761000},"page":"1043-1055","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["TTS-Portuguese Corpus: a corpus for speech synthesis in Brazilian Portuguese"],"prefix":"10.1007","volume":"56","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0160-7173","authenticated-orcid":false,"given":"Edresson","family":"Casanova","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5647-0891","authenticated-orcid":false,"given":"Arnaldo Candido","family":"Junior","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9637-9657","authenticated-orcid":false,"given":"Christopher","family":"Shulby","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5885-6747","authenticated-orcid":false,"given":"Frederico Santos de","family":"Oliveira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6679-5702","authenticated-orcid":false,"given":"Jo\u00e3o Paulo","family":"Teixeira","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2059-9463","authenticated-orcid":false,"given":"Moacir Antonelli","family":"Ponti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5108-2630","authenticated-orcid":false,"given":"Sandra","family":"Alu\u00edsio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,12]]},"reference":[{"key":"9570_CR1","doi-asserted-by":"crossref","unstructured":"Alencar, V., & Alcaim, A. (2008). LSF and lPC-derived features for large vocabulary distributed continuous speech recognition in Brazilian Portuguese. In 2008 42nd\nAsilomar conference on signals, systems and computers (pp. 1237\u20131241). IEEE.","DOI":"10.1109\/ACSSC.2008.5074614"},{"key":"9570_CR2","unstructured":"Arik, S. O., Chrzanowski, M., Coates, A., Diamos, G., Gibiansky, A., Kang, Y., Li, X., Miller, J., Raiman, J., & Sengupta, S., & Ng, A. (2017). Deep voice: Real-time neural text-to-speech. arXiv preprint. http:\/\/arxiv.org\/abs\/170207825"},{"key":"9570_CR3","unstructured":"Ar\u0131k, S. O., Diamos, G., Gibiansky, A., Miller, J., Peng, K., Ping, W., Raiman, J., & Zhou, Y. (2017). Deep voice 2: Multi-speaker neural text-to-speech. arXiv preprint. http:\/\/arxiv.org\/abs\/170508947"},{"key":"9570_CR4","doi-asserted-by":"crossref","unstructured":"Aroon, A., & Dhonde, S. (2015). Statistical parametric speech synthesis: A review. In 2015 IEEE 9th international conference on intelligent systems and control (ISCO) (pp. 1\u20135). IEEE.","DOI":"10.1109\/ISCO.2015.7282379"},{"key":"9570_CR5","unstructured":"Ba, J. L., Kiros, J. R., & Hinton, G. E. (2016). Layer normalization. arXiv preprint. http:\/\/arxiv.org\/abs\/160706450"},{"key":"9570_CR6","unstructured":"Bahdanau, D., Cho, K., & Bengio, Y. (2014). Neural machine translation by jointly learning to align and translate. arXiv preprint. http:\/\/arxiv.org\/abs\/14090473"},{"key":"9570_CR7","doi-asserted-by":"crossref","unstructured":"Benesty, J., Chen, J., & Habets, E. A. (2011). Speech enhancement in the STFT domain. Springer Science & Business Media.","DOI":"10.1007\/978-3-642-23250-3"},{"key":"9570_CR8","doi-asserted-by":"crossref","unstructured":"Braude, D. A., Shimodaira, H., & Youssef, A. B. (2013). Template-warping based speech driven head motion synthesis. In Interspeech (pp. 2763\u20132767).","DOI":"10.21437\/Interspeech.2013-633"},{"key":"9570_CR9","doi-asserted-by":"crossref","unstructured":"Charpentier, F., & Stella, M. (1986). Diphone synthesis using an overlap-add technique for speech waveforms concatenation. In ICASSP\u201986. IEEE international conference\non acoustics, speech, and signal processing (Vol.\u00a011, pp. 2015\u20132018). IEEE.","DOI":"10.1109\/ICASSP.1986.1168657"},{"key":"9570_CR10","doi-asserted-by":"crossref","unstructured":"Cho, K., Van Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., & Bengio, Y. (2014). Learning phrase representations using RNN encoder\u2013decoder for statistical machine translation. arXiv preprint. http:\/\/arxiv.org\/abs\/14061078","DOI":"10.3115\/v1\/D14-1179"},{"key":"9570_CR11","unstructured":"Chung, J., Gulcehre, C., Cho, K., & Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint. http:\/\/arxiv.org\/abs\/14123555"},{"issue":"3","key":"9570_CR12","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1049\/et.2017.0330","volume":"12","author":"P Dempsey","year":"2017","unstructured":"Dempsey, P. (2017). The teardown: Google home personal assistant. Engineering & Technology, 12(3), 80\u201381.","journal-title":"Engineering & Technology"},{"key":"9570_CR13","unstructured":"Durkan, C., Bekasov, A., Murray, I., & Papamakarios, G. (2019). Neural spline flows. In Advances in neural information processing systems (pp. 7511\u20137522)"},{"key":"9570_CR14","unstructured":"Goodfellow, I., Bengio, Y., Courville, A., & Bengio, Y. (2016). Deep learning (Vol.\u00a01). MIT Press."},{"issue":"2","key":"9570_CR15","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., & Lim, J. (1984). Signal estimation from modified short-time Fourier transform. IEEE Transactions on Acoustics, Speech, and Signal Processing, 32(2), 236\u2013243.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9570_CR16","unstructured":"Gruber, T. R. (2009) Siri, a virtual personal assistant-bringing intelligence to the interface. In Semantic technologies conference."},{"key":"9570_CR17","unstructured":"G\u00f6lge, E. (2019). Deep learning for text to speech. https:\/\/github.com\/mozilla\/TTS"},{"key":"9570_CR18","unstructured":"Hoogeboom, E., Van Den Berg, R., & Welling, M. (2019). Emerging convolutions for generative normalizing flows. arXiv preprint. http:\/\/arxiv.org\/abs\/190111137"},{"key":"9570_CR19","unstructured":"Ito, K. (2017). The lj speech dataset. Retrieved April 29, 2020, from https:\/\/keithito.com\/LJ-Speech-Dataset\/"},{"key":"9570_CR20","unstructured":"Kalchbrenner, N., Espeholt, L., Simonyan, K., van den Oord, A., Graves, A., & Kavukcuoglu, K. (2016). Neural machine translation in linear time. arXiv preprint. http:\/\/arxiv.org\/abs\/161010099"},{"key":"9570_CR21","unstructured":"Kim, J., Kim, S., Kong, J., & Yoon, S. (2020). Glow-TTS: A generative flow for text-to-speech via monotonic alignment search. arXiv preprint. http:\/\/arxiv.org\/abs\/200511129"},{"key":"9570_CR22","unstructured":"Kingma, D. P., Salimans, T., Jozefowicz, R., Chen, X., Sutskever, I., & Welling, M. (2016). Improved variational inference with inverse autoregressive flow. In Advances in neural information processing systems (pp 4743\u20134751)."},{"key":"9570_CR23","doi-asserted-by":"crossref","unstructured":"Klatt, D. H. (1980) Software for a cascade\/parallel formant synthesizer. The Journal of the Acoustical Society of America, 67(3), 971\u2013995.","DOI":"10.1121\/1.383940"},{"key":"9570_CR24","unstructured":"Kumar, R., Kumar, K., Anand, V., Bengio, Y., & Courville, A. (2020) NU-GAN: High resolution neural upsampling with GAN. arXiv preprint. http:\/\/arxiv.org\/abs\/201011362"},{"key":"9570_CR25","unstructured":"Mehri, S., Kumar, K., Gulrajani, I., Kumar, R., Jain, S., Sotelo, J., Courville, A., Bengio, Y., & Sample R. N. N. (2017). Char2wav: End-to-end speech synthesis. In International conference on learning representations, workshop."},{"key":"9570_CR26","doi-asserted-by":"crossref","unstructured":"Miao, C., Liang, S., Chen, M., Ma, J., Wang, S., & Xiao, J. (2020). Flow-TTS: A non-autoregressive network for text to speech based on flow. In ICASSP 2020\u20132020 IEEE international conference on acoustics, speech and signal processing\n(ICASSP) (pp. 7209\u20137213). IEEE.","DOI":"10.1109\/ICASSP40776.2020.9054484"},{"issue":"7","key":"9570_CR27","doi-asserted-by":"publisher","first-page":"1877","DOI":"10.1587\/transinf.2015EDP7457","volume":"99","author":"M Morise","year":"2016","unstructured":"Morise, M., Yokomori, F., & Ozawa, K. (2016). World: A vocoder-based high-quality speech synthesis system for real-time applications. IEICE Transactions on Information and Systems, 99(7), 1877\u20131884.","journal-title":"IEICE TRANSACTIONS on Information and Systems"},{"key":"9570_CR28","unstructured":"Park, K. (2018). A tensorflow implementation of DC-TTS. https:\/\/github.com\/Kyubyong\/dc_tts"},{"key":"9570_CR29","unstructured":"Ping, W., Peng, K., Gibiansky, A., Arik, S. O., Kannan, A., Narang, S., Raiman, J., & Miller, J. (2017). Deep voice 3: 2000-speaker neural text-to-speech. arXiv preprint. http:\/\/arxiv.org\/abs\/171007654"},{"key":"9570_CR30","doi-asserted-by":"crossref","unstructured":"Pratap, V., Xu, Q., Sriram, A., Synnaeve, G., & Collobert, R. (2020). Mls: A large-scale multilingual dataset for speech research. In Proceedings of Interspeech 2020 (pp. 2757\u20132761).","DOI":"10.21437\/Interspeech.2020-2826"},{"key":"9570_CR31","doi-asserted-by":"crossref","unstructured":"Purington, A., Taft, J. G., Sannon, S., Bazarova, N. N., & Taylor, S. H (2017). \u201cAlexa is my new BFF\u201d: Social roles, user satisfaction, and personification of the Amazon echo. In Proceedings of the 2017 CHI conference extended abstracts on human factors in computing systems (pp. 2853\u20132859).","DOI":"10.1145\/3027063.3053246"},{"issue":"1","key":"9570_CR32","doi-asserted-by":"publisher","first-page":"230","DOI":"10.14209\/jcis.2020.25","volume":"35","author":"IM Quintanilha","year":"2020","unstructured":"Quintanilha, I. M., Netto, S. L., & Biscainho, L. W. P. (2020). An open-source end-to-end ASR system for Brazilian Portuguese using DNNs built from newly assembled corpora. Journal of Communication and Information Systems, 35(1), 230\u2013242.","journal-title":"Journal of Communication and Information Systems"},{"key":"9570_CR33","doi-asserted-by":"crossref","unstructured":"Quintas, S., & Trancoso, I. (2020). Evaluation of deep learning approaches to text-to-speech systems for European Portuguese. In International conference on computational processing of the Portuguese language (pp. 34\u201342). Springer.","DOI":"10.1007\/978-3-030-41505-1_4"},{"key":"9570_CR34","doi-asserted-by":"crossref","unstructured":"Ribeiro, F., Flor\u00eancio, D., Zhang, C., & Seltzer, M. (2011). Crowdmos: An approach for crowdsourcing mean opinion score studies. In 2011 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 2416\u20132419). IEEE.","DOI":"10.1109\/ICASSP.2011.5946971"},{"key":"9570_CR35","unstructured":"Seara, I. (1994). Estudo estat\u00edstico dos fonemas do portugu\u00eas brasileiro falado na capital de santa catarina para elabora\u00e7\u00e3o de frases foneticamente balanceadas. PhD thesis, Disserta\u00e7\u00e3o de Mestrado, Universidade Federal de Santa Catarina\u00a0..."},{"key":"9570_CR36","doi-asserted-by":"crossref","unstructured":"Shen, J., Pang, R., Weiss, R. J., Schuster, M., Jaitly, N., Yang, Z., Chen, Z., Zhang, Y., Wang, Y., Skerrv-Ryan, R., & Saurous R. A. (2018). Natural TTS synthesis by conditioning WaveNet on Mel spectrogram predictions. In 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 4779\u20134783). IEEE.","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"9570_CR37","doi-asserted-by":"crossref","unstructured":"Siddhi, D., Verghese, J. M., & Bhavik, D. (2017). Survey on various methods of text to speech synthesis. International Journal of Computer Applications, 165(6), 26\u201330.","DOI":"10.5120\/ijca2017913891"},{"key":"9570_CR100","unstructured":"Sotelo, J., Mehri, S., Kumar, K., Santos, J. F., Kastner, K., Courville, A., & Bengio, Y. (2017). Char2wav: End-to-end speech synthesis. In International conference on learning representations, workshop."},{"key":"9570_CR38","unstructured":"Srivastava, R. K., Greff, K., & Schmidhuber, J. (2015). Training very deep networks. In Advances in neural information processing systems (pp. 2377\u20132385)."},{"key":"9570_CR39","doi-asserted-by":"crossref","unstructured":"Tachibana, H., Uenoyama, K., & Aihara, S. (2017). Efficiently trainable text-to-speech system based on deep convolutional networks with guided attention. arXiv preprint. http:\/\/arxiv.org\/abs\/171008969","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"9570_CR40","doi-asserted-by":"crossref","unstructured":"Tamamori, A., Hayashi, T., Kobayashi, K., Takeda, K., & Toda, T. (2017). Speaker-dependent WaveNet vocoder. In Proceedings of Interspeech (pp. 1118\u20131122).","DOI":"10.21437\/Interspeech.2017-314"},{"key":"9570_CR41","doi-asserted-by":"crossref","unstructured":"Teixeira, J. P., Freitas, D., Braga, D., Barros, M. J., & Latsch, V. (2001). Phonetic events from the labeling the European Portuguese database for speech synthesis, FEUP\/IPBDB. In Seventh European conference on speech communication and technology.","DOI":"10.21437\/Eurospeech.2001-400"},{"key":"9570_CR42","doi-asserted-by":"crossref","unstructured":"Teixeira, J. P., Freitas, D., & Fujisaki, H. (2003). Prediction of Fujisaki model\u2019s phrase commands. In Eighth European conference on speech communication and technology.","DOI":"10.21437\/Eurospeech.2003-154"},{"key":"9570_CR43","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Yoshimura, T., Masuko, T., Kobayashi, T., & Kitamura, T. (2000). Speech parameter generation algorithms for HMM-based speech synthesis. In 2000 IEEE international conference on acoustics, speech, and signal processing. Proceedings (Cat. No. 00CH37100) (Vol.\u00a03, pp. 1315\u20131318). IEEE.","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"9570_CR44","doi-asserted-by":"crossref","unstructured":"Valin, J. M. (2017). A hybrid DSP\/deep learning approach to real-time full-band speech enhancement. arXiv preprint. http:\/\/arxiv.org\/abs\/170908243","DOI":"10.1109\/MMSP.2018.8547084"},{"key":"9570_CR45","unstructured":"Valle, R., Shih, K., Prenger, R., & Catanzaro, B. (2020). Flowtron: An autoregressive flow-based generative network for text-to-speech synthesis. arXiv preprint. http:\/\/arxiv.org\/abs\/200505957"},{"key":"9570_CR46","unstructured":"Van Den Oord, A., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A., & Kavukcuoglu, K. (2016). Wavenet: A generative model for raw audio. arXiv preprint. http:\/\/arxiv.org\/abs\/160903499"},{"key":"9570_CR47","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. In Advances in neural information processing systems (pp. 5998\u20136008)."},{"key":"9570_CR48","doi-asserted-by":"crossref","unstructured":"Wang, W. Y., & Georgila, K. (2011). Automatic detection of unnatural word-level segments in unit-selection speech synthesis. In 2011 IEEE workshop on automatic speech recognition & understanding (pp. 289\u2013294). IEEE.","DOI":"10.1109\/ASRU.2011.6163946"},{"key":"9570_CR49","unstructured":"Wang, Y., Skerry-Ryan, R., Stanton, D., Wu, Y., Weiss, R. J., Jaitly, N., Yang, Z., Xiao, Y., Chen, Z., & Bengio, S., & Le, Q. V. (2017). Tacotron: A fully end-to-end text-to-speech synthesis model. arXiv preprint. http:\/\/arxiv.org\/abs\/170310135"},{"key":"9570_CR50","unstructured":"Yu, F., & Koltun, V. (2015). Multi-scale context aggregation by dilated convolutions. arXiv preprint. http:\/\/arxiv.org\/abs\/151107122"},{"key":"9570_CR51","doi-asserted-by":"crossref","unstructured":"Ze, H., Senior, A., & Schuster, M. (2013). Statistical parametric speech synthesis using deep neural networks. In 2013 IEEE international conference on acoustics, speech and signal processing (pp. 7962\u20137966). IEEE.","DOI":"10.1109\/ICASSP.2013.6639215"},{"issue":"5","key":"9570_CR52","doi-asserted-by":"publisher","first-page":"1645","DOI":"10.1109\/TASL.2007.899236","volume":"15","author":"X Zhu","year":"2007","unstructured":"Zhu, X., Beauregard, G. T., & Wyse, L. L. (2007). Real-time signal estimation from modified short-time fourier transform magnitude spectra. IEEE Transactions on Audio, Speech, and Language Processing, 15(5), 1645\u20131653.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"}],"container-title":["Language Resources and Evaluation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-021-09570-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10579-021-09570-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-021-09570-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,18]],"date-time":"2022-08-18T10:07:19Z","timestamp":1660817239000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10579-021-09570-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1,12]]},"references-count":53,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,9]]}},"alternative-id":["9570"],"URL":"https:\/\/doi.org\/10.1007\/s10579-021-09570-4","relation":{},"ISSN":["1574-020X","1574-0218"],"issn-type":[{"type":"print","value":"1574-020X"},{"type":"electronic","value":"1574-0218"}],"subject":[],"published":{"date-parts":[[2022,1,12]]},"assertion":[{"value":"23 November 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 January 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflicts of interest to declare.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}