{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T14:12:34Z","timestamp":1774879954534,"version":"3.50.1"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T00:00:00Z","timestamp":1767916800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T00:00:00Z","timestamp":1767916800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s10772-025-10241-w","type":"journal-article","created":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T10:09:37Z","timestamp":1767953377000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["STFT-GradTTS: a robust, diffusion-based speech synthesis system with iSTFT decoder for Bangla"],"prefix":"10.1007","volume":"29","author":[{"given":"Mushahid","family":"Intesum","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abdullah Ibne","family":"Masud","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Md. Rezaul","family":"Karim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Md. Ashraful","family":"Islam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,9]]},"reference":[{"key":"10241_CR1","doi-asserted-by":"publisher","unstructured":"Ahmed, K., Mandal, P., & Hossain, B. M. M. (2019). Text to speech synthesis for Bangla language. International Journal of Information Engineering and Electronic Business, 11, 1\u20139. https:\/\/doi.org\/10.5815\/ijieeb.2019.02.01","DOI":"10.5815\/ijieeb.2019.02.01"},{"key":"10241_CR2","unstructured":"Anastassiou, P., Chen, J., Chen, J., Chen, Y., Chen, Z., Chen, Z., Cong, J., Deng, L., Ding, C., & Gao, L., et al. (2024). Seed- TTS: A family of highquality versatile speech generation models. arXiv preprint arXiv:2406.02430."},{"key":"10241_CR3","unstructured":"Arik, S. O., Chrzanowski, M., Coates, A., Diamos, G., Gibiansky, A., Kang, Y., Li, X., Miller, J., Ng, A., Raiman, J., Sengupta, S., & Shoeybi, M. (2017). Deep Voice: Real-time neural text-to-speech."},{"key":"10241_CR4","doi-asserted-by":"crossref","unstructured":"Baum, L. E., & Petrie, T. (1966). Statistical inference for probabilistic functions of finite state markov chains. The Annals of Mathematical Statistics, 37(6), 1554\u20131563.","DOI":"10.1214\/aoms\/1177699147"},{"key":"10241_CR5","doi-asserted-by":"crossref","unstructured":"Chinen, M., Lim, F. S., Skoglund, J., Gureev, N., O\u2019Gorman, F., & Hines, A. (2020). Visqol v3: An open source production ready objective speech and audio metric. In 2020 twelfth international conference on quality of multimedia experience (QoMEX) (pp. 1\u20136). IEEE.","DOI":"10.1109\/QoMEX48832.2020.9123150"},{"key":"10241_CR6","doi-asserted-by":"crossref","unstructured":"Choi, S., Han, S., Kim, D., & Ha, S. (2020). Attentron: Few-shot text-to-speech utilizing attention-based variable-length embedding. In Interspeech 2020.","DOI":"10.21437\/Interspeech.2020-2096"},{"key":"10241_CR7","unstructured":"D\u00e9fossez, A., Copet, J., Synnaeve, G., & Adi, Y. (2022). High fidelity neural audio compression. OpenReview.net"},{"key":"10241_CR8","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.-W., Lee, K., & Toutanova, K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of NAACL-HLT 2019 (pp. 4171\u2013 4186). Association for Computational Linguistics.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10241_CR9","doi-asserted-by":"crossref","unstructured":"Eskimez, S. E., Wang, X., Thakker, M., Li, C., Tsai, C.-H., Xiao, Z., Yang, H., Zhu, Z., Tang, M., Tan, X., Liu, Y., Zhao, S., & Kanda, N. (2024). E2 TTS: Embarrassingly easy fully non-autoregressive zero-shot TTS. In 2024 IEEE spoken language technology workshop (SLT) (pp. 682\u2013689). IEEE.","DOI":"10.1109\/SLT61566.2024.10832320"},{"key":"10241_CR10","doi-asserted-by":"crossref","unstructured":"Gao, Y., Morioka, N., Zhang, Y., & Chen, N. (2023). E3 TTS: Easy end-to-end diffusion-based text to speech. arXiv:2311.00945v1 [cs.SD].","DOI":"10.1109\/ASRU57964.2023.10389766"},{"issue":"2","key":"10241_CR11","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D. Griffin","year":"1984","unstructured":"Griffin, D., & Lim, J. (1984). Signal estimation from modified short-time fourier transform. IEEE Transactions on Acoustics, Speech, and Signal Processing, 32(2), 236\u2013243. https:\/\/doi.org\/10.1109\/TASSP.1984.1164317","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"10241_CR12","unstructured":"Gutkin, A., Ha, L., Jansche, M., Pipatsrisawat, K., & Sproat, R. (2016). TTS for low resource languages: A Bangla synthesizer. In Proceedings of the tenth international conference on language resources and evaluation (LREC\u201916) (pp. 2005\u20132010). European Language Resources Association (ELRA), Portoro\u017e, Slovenia. https:\/\/aclanthology.org\/L16-1317"},{"key":"10241_CR13","unstructured":"Hernandez-Mena, C. D. (2019). Tedx Spanish corpus. Audio and transcripts in Spanish taken from the TEDx Talks; shared under the cc BY-NC-ND 4.0 license. Web Download."},{"key":"10241_CR14","doi-asserted-by":"publisher","unstructured":"Islam Pial, T., Salim Aunti, S., Ahmed, S., & Heickal, H. (2018). End-to-end speech synthesi for Bangla with text normalization. In End-to-end speech synthesis for Bangla with text normalization, 2018 5th international conference on computational science\/ intelligence and applied informatics (CSII) (Vol. 1(1), (pp. 66\u201371). Yonago, Japan. https:\/\/doi.org\/10.1109\/CSII.2018.00019","DOI":"10.1109\/CSII.2018.00019"},{"key":"10241_CR15","unstructured":"Ito, K., & Johnson, L. (2017). The LJ speech dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/"},{"key":"10241_CR16","doi-asserted-by":"crossref","unstructured":"Kawamura, M., Shirahata, Y., Yamamoto, R., & Tachibana, K. (2023). Lightweight and high-fidelity end-to-end text-to-speech with multi-band generation and inverse short-time Fourier transform. https:\/\/arxiv.org\/abs\/2210.15975","DOI":"10.1109\/ICASSP49357.2023.10095296"},{"key":"10241_CR17","unstructured":"Kim, H., Kim, S., & Yoon, S. (2022). Guided-TTS: A diffusion model for text-to-speech via Classifier guidance. In Proceedings of the 39th international conference on machine learning."},{"key":"10241_CR18","unstructured":"Kim, J., Kim, S., Kong, J., & Yoon, S. (2020). Glow-TTS: A generative flow for text-to-speech via monotonic alignment search. NISP'20: Proceedings of the 34th International Conference on Neural Information Processing Systems, 8067\u20138077."},{"key":"10241_CR19","unstructured":"Kim, J., Kong, J., & Son, J. (2021). Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In Proceedings of the 38th international conference on machine learning."},{"key":"10241_CR20","unstructured":"Kim, J., Lee, K., Chung, S., & Cho, J. (2024). CLaM-TTS: Improving neural codec language model for zero-shot text-to-speech. arXiv preprint arXiv:2404.02781."},{"key":"10241_CR21","unstructured":"Kingma, D. P., & Ba, J. (2014). Adam: A method for stochastic optimization. In International conference on learning representations."},{"key":"10241_CR22","unstructured":"Lajszczak, M., C\u00e1mbara, G., Li, Y., Beyhan, F., Van Korlaar, A., Yang, F., Joly, A., Mart\u00edn-Cortinas, \u00c1., Abbas, A., Michalski, A., Moinet, A., Karlapati, S., Muszynska, E., Guo, H., Putrycz, B., L\u00f3pez, S., Yoo, K., Sokolova, E., & Drugman, T. (2025). BASE TTS: Lessons from building a billion-parameter text-to-speech model on 100k hours of data. arXiv preprint arXiv:2402.08093."},{"key":"10241_CR23","doi-asserted-by":"publisher","unstructured":"Li,N., Liu, S., Liu, Y., Zhao, S., & Liu, M. (2019). Neural speech synthesis with transformer network. Proceedings of the AAAI Conference on Artificial Intelligence, 33(1), 6706\u20136713. https:\/\/doi.org\/10.1609\/aaai.v33i01.33016706","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"10241_CR24","doi-asserted-by":"publisher","first-page":"1877","DOI":"10.1587\/transinf.2015EDP7457","volume":"E99","author":"M. Morise","year":"2016","unstructured":"Morise, M., Yokomori, F., & Ozawa, K. (2016). World: A vocoder-based high-quality speech synthesis system for real-time applications. IEICE Transactions on Information and Systems, E99.D(7), 1877\u20131884. https:\/\/doi.org\/10.1587\/transinf.2015EDP7457","journal-title":"IEICE Transactions on Information and Systems"},{"key":"10241_CR25","doi-asserted-by":"publisher","unstructured":"Okamoto, T., Toda, T., & Kawai, H. (2021). Multistream hifi-gan with data-driven waveform decomposition. In 2021 IEEE automatic speech recognition and understanding workshop (ASRU) (pp. 610\u2013617). https:\/\/doi.org\/10.1109\/ASRU51503.2021.9688194","DOI":"10.1109\/ASRU51503.2021.9688194"},{"key":"10241_CR26","unstructured":"Oord, A., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A., & Kavukcuoglu, K. (2016). WaveNet: A generative model for raw audio. arXiv:1609.03499v2 [cs.SD]"},{"key":"10241_CR27","doi-asserted-by":"publisher","unstructured":"Panayotov, V., Chen, G., Povey, D., & Khudanpur, S. (2015). Librispeech: An ASR corpus based on public domain audio books. https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"10241_CR28","unstructured":"Ping, W., Peng, K., Gibiansky, A., Arik, S. O., Kannan, A., Narang, S., Raiman, J., & Miller, J. (2018). Deep voice 3: Scaling text-to-speech with convolutional sequence learning. ICLR 2018 Conference paper."},{"key":"10241_CR29","unstructured":"Popov, V., Vovk, I., Gogoryan, V., Sadekova, T., & Kudinov, M. (2021). Grad-TTS: A diffusion probabilistic model for text-to-speech. In Proceedings of the 38th international conference on machine learning."},{"key":"10241_CR30","doi-asserted-by":"crossref","unstructured":"Pratap, V., Xu, Q., Sriram, A., Synnaeve, G., & Collobert, R. (2020). MLS: A large-scale multilingual dataset for speech research. arXiv:2012.03411v2 [eess.AS]","DOI":"10.21437\/Interspeech.2020-2826"},{"key":"10241_CR31","unstructured":"Ren, Y., Hu, C., Tan, X., Qin, T., Zhao, S., Zhao, Z., & Liu, T.-Y. (2022). FastSpeech 2: Fast and high-quality end-to-end text to speech."},{"key":"10241_CR32","unstructured":"Ren, Y., Ruan, Y., Tan, X., Qin, T., Zhao, S., Zhao, Z., & Liu, T.-Y. (2019). FastSpeech: Fast, robust and controllable text to speech.In Proceedings of the 33rd international conference on neural information processing systems (pp. 3171\u2013 3180)."},{"key":"10241_CR33","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). Unet: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention (MICCAI 2015) (pp. 234\u2013241). Springer.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"10241_CR34","unstructured":"Shah, N., Tambrahalli, V., Kosgi, S., Pedanekar, N., & Gandhi, V. (2023). MParrotTTS: Multilingual Multi-speaker text to speech synthesis in low resource setting. arXiv:2305.11926v1 [cs.SD]"},{"key":"10241_CR35","doi-asserted-by":"crossref","unstructured":"Shen, J., Pang, R., Weiss, R. J., Schuster, M., Jaitly, N., Yang, Z., Chen, Z., Zhang, Y., Wang, Y., Skerry-Ryan, R., Saurous, R. A., Agiomyrgiannakis, Y., & Wu, Y. (2018). Natural TTS synthesis by conditioning WaveNet on MEL spectrogram predictions. In 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP 2018). IEEE.","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"10241_CR36","unstructured":"Shen, K., Ju, Z., Tan, X., Liu, Y., Leng, Y., He, L., Qin, T., Zhao, S., & Bian, J. (2023). NaturalSpeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers. In The twelfth international conference on learning representations (ICLR 2024)."},{"key":"10241_CR37","doi-asserted-by":"publisher","unstructured":"Sodimana,K., Pipatsrisawat, K., Ha, L., Jansche, M., Kjartansson, O., Silva, P. D., & Sarin, S. (2018). A step-by-step process for building TTS voices using open source data and framework for Bangla, Javanese, Khmer, Nepali, Sinhala, and Sundanese. In 6th Workshop on spoken language technologies for under-resourced languages (SLTU 2018), 29-31 August 2018, Gurugram, India. https:\/\/doi.org\/10.21437\/SLTU.2018-14","DOI":"10.21437\/SLTU.2018-14"},{"key":"10241_CR38","unstructured":"Wang, C., Chen, S., Wu, Y., Zhang, Z., Zhou, L., Liu, S., Chen, Z., Liu, Y., Wang, H., Li, J., He, L., Zhao, S., & Wei, F. (2023). Neural codec language models are zero-shot text to speech synthesizers. arXiv:2301.02111v1 [cs.CL]"},{"key":"10241_CR39","unstructured":"Wang, X., Jiang, M., Ma, Z., Zhang, Z., Liu, S., Li, L., Liang, Z., Zheng, Q., Wang, R., & Feng, X., et al. (2025). Spark-TTS: An efficient LLM-based text-to-speech model with single-stream decoupled speech tokens. arXiv preprint arXiv:2503.01710."},{"key":"10241_CR40","doi-asserted-by":"crossref","unstructured":"Wang, Y., Skerry-Ryan, R., Stanton, D., Wu, Y., Weiss, R. J., Jaitly, N., Yang, Z., Xiao, Y., Chen, Z., Bengio, S., & Le, Q., Y., Clark, & R. (2017). Agiomyrgiannakis. Saurous, R.A: Tacotron: Towards End-to-End Speech Synthesis.","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"10241_CR41","unstructured":"Yamagishi, J., Veaux, C., & MacDonald, K. (2019). CSTR VCTK corpus: English Multi-speaker corpus for CSTR voice cloning toolkit (version 0.92)."},{"key":"10241_CR42","doi-asserted-by":"crossref","unstructured":"Zeghidour, N., Luebs, A., Omran, A., Skoglund, J., & Tagliasacchi, M. (2021). SoundStream: An end-to-end neural audio codec. IEEE\/ACM Transactions on Audio, Speech and Language Processing, 30","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"10241_CR43","doi-asserted-by":"crossref","unstructured":"Zhizheng Wu, Watts, O., & King, S. (2016). Merlin: An open source neural network speech synthesis system. In 9th ISCA speech synthesis workshop.","DOI":"10.21437\/SSW.2016-33"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-025-10241-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-025-10241-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-025-10241-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T13:23:11Z","timestamp":1774876991000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-025-10241-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,9]]},"references-count":43,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["10241"],"URL":"https:\/\/doi.org\/10.1007\/s10772-025-10241-w","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,9]]},"assertion":[{"value":"30 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"20"}}