{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T09:08:47Z","timestamp":1776848927204,"version":"3.51.2"},"reference-count":94,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100000646","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP23K24870"],"award-info":[{"award-number":["JP23K24870"]}],"id":[{"id":"10.13039\/501100000646","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000646","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP24K02959"],"award-info":[{"award-number":["JP24K02959"]}],"id":[{"id":"10.13039\/501100000646","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000646","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP24K21322"],"award-info":[{"award-number":["JP24K21322"]}],"id":[{"id":"10.13039\/501100000646","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000646","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP24H00741"],"award-info":[{"award-number":["JP24H00741"]}],"id":[{"id":"10.13039\/501100000646","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000646","name":"Japan Society for the Promotion of Science (JSPS) KAKENHI","doi-asserted-by":"publisher","award":["JP24H00170"],"award-info":[{"award-number":["JP24H00170"]}],"id":[{"id":"10.13039\/501100000646","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012389","name":"Commissioned Research of National Institute of Information and Communications Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012389","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/access.2026.3683761","type":"journal-article","created":{"date-parts":[[2026,4,21]],"date-time":"2026-04-21T19:50:21Z","timestamp":1776801021000},"page":"58495-58514","source":"Crossref","is-referenced-by-count":0,"title":["Deep Hidden Semi-Markov Model-Based Speech Synthesis"],"prefix":"10.1109","volume":"14","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-3978-5130","authenticated-orcid":false,"given":"Yoshihiko","family":"Nankaku","sequence":"first","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5390-3701","authenticated-orcid":false,"given":"Takato","family":"Fujimoto","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takenori","family":"Yoshimura","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Takaki","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2081-0396","authenticated-orcid":false,"given":"Kei","family":"Hashimoto","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7611-9435","authenticated-orcid":false,"given":"Keiichiro","family":"Oura","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6143-0133","authenticated-orcid":false,"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref2","article-title":"A survey on neural speech synthesis","author":"Tan","year":"2021","journal-title":"arXiv:2106.15561"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/5.18626"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2009.11.011"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.825"},{"key":"ref7","article-title":"Neural machine translation by jointly learning to align and translate","author":"Bahdanau","year":"2014","journal-title":"arXiv:1409.0473"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1972"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.31390\/gradschool_dissertations.4601"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054106"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2021-35"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053546"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746686"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2968"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6337"},{"key":"ref20","article-title":"Non-attentive tacotron: Robust and controllable neural TTS synthesis including unsupervised duration modeling","author":"Shen","year":"2020","journal-title":"arXiv:2010.04301"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414718"},{"key":"ref22","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Ren"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref24","first-page":"195","article-title":"Deep voice: Real-time neural text-to-speech","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Ar\u0131k"},{"key":"ref25","article-title":"TalkNet: Fully-convolutional non-autoregressive speech synthesis model","author":"Beliaev","year":"2020","journal-title":"arXiv:2005.05514"},{"key":"ref26","first-page":"3171","article-title":"FastSpeech: Fast, robust and controllable text to speech","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ren"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2867"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054119"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2123"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414499"},{"key":"ref31","first-page":"7700","article-title":"EfficientTTS: An efficient and high-quality text-to-speech architecture","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Miao"},{"key":"ref32","first-page":"8067","article-title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref33","article-title":"End-to-end adversarial text-to-speech","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Donahue"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1461"},{"key":"ref35","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Kim"},{"key":"ref36","article-title":"NaturalSpeech: End-to-end text to speech synthesis with human-level quality","author":"Tan","year":"2022","journal-title":"arXiv:2205.04421"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746158"},{"key":"ref38","article-title":"Auto-encoding variational Bayes","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1113"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683623"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2477"},{"key":"ref42","article-title":"Hierarchical generative modeling for controllable speech synthesis","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hsu"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-259"},{"key":"ref44","article-title":"Bidirectional variational inference for non-autoregressive text-to-speech","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Lee"},{"key":"ref45","article-title":"VARA-TTS: Non-autoregressive text-to-speech synthesis based on very deep VAE with residual attention","author":"Liu","year":"2021","journal-title":"arXiv:2102.06431"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095100"},{"key":"ref47","first-page":"1530","article-title":"Variational inference with normalizing flows","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","volume":"37","author":"Rezende"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054484"},{"key":"ref49","article-title":"Flowtron: An autoregressive flow-based generative network for text-to-speech synthesis","author":"Valle","year":"2020","journal-title":"arXiv:2005.05957"},{"key":"ref50","article-title":"Mixture density networks","author":"Bishop","year":"1994"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854321"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-18"},{"key":"ref53","first-page":"3581","article-title":"Semi-supervised learning with deep generative models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"27","author":"Kingma"},{"key":"ref54","first-page":"1278","article-title":"Stochastic backpropagation and approximate inference in deep generative models","volume-title":"Proc. 31st Int. Conf. Mach. Learn.","author":"Rezende"},{"key":"ref55","first-page":"2954","article-title":"Composing graphical models with neural networks for structured representations and fast inference","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Johnson"},{"key":"ref56","article-title":"Deep unsupervised clustering with Gaussian mixture variational autoencoders","author":"Dilokthanakul","year":"2016","journal-title":"arXiv:1611.02648"},{"key":"ref57","article-title":"Variational deep embedding: An unsupervised and generative approach to clustering","author":"Jiang","year":"2016","journal-title":"arXiv:1611.05148"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1160"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2021.103347"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967987"},{"key":"ref61","article-title":"Recurrent hidden semi-Markov model","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Dai"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/339"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref64","article-title":"On structured variational approximations","author":"Ghahramani","year":"2002"},{"key":"ref65","article-title":"Neural sequence-to-sequence speech synthesis using a hidden semi-Markov model based structured attention mechanism","author":"Nankaku","year":"2021","journal-title":"arXiv:2108.13985"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.2307\/2984875"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1016\/S0885-2308(86)80009-2"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/89.388149"},{"key":"ref69","article-title":"Implementing an HSMM-based speech synthesis system using an efficient forward-backward algorithm","author":"Zen","year":"2007"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2006.01.002"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/MLSP.2015.7324368"},{"key":"ref73","article-title":"The use of context in large vocabulary speech recognition","author":"Odell","year":"1995"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1997-52"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1999-513"},{"key":"ref76","article-title":"A structured variational auto-encoder for learning deep hierarchies of sparse features","author":"Salimans","year":"2016","journal-title":"arXiv:1602.08734"},{"key":"ref77","article-title":"Ladder variational autoencoders","author":"Sonderby","year":"2016","journal-title":"arXiv:1602.02282"},{"key":"ref78","article-title":"Structured attention networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kim"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007665907178"},{"key":"ref80","first-page":"6309","article-title":"Neural discrete representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"van den Oord"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1967.1054010"},{"key":"ref82","first-page":"179","article-title":"XIMERA: A new TTS from ATR based on corpus-based technologies","volume-title":"Proc. 5th ISCA Speech Synth. Workshop","author":"Kawai"},{"key":"ref83","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"},{"key":"ref84","article-title":"WaveGrad: Estimating gradients for waveform generation","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Chen"},{"key":"ref85","first-page":"17022","article-title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kong"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-439"},{"key":"ref87","volume-title":"Fastspeech 2-PyTorch Implementation","author":"Chien","year":"2021"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"ref89","volume-title":"Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search","author":"Kim","year":"2020"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10448291"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054466"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"ref94","volume-title":"The LJ Speech Dataset","author":"Ito","year":"2017"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/11323511\/11481063.pdf?arnumber=11481063","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T08:10:49Z","timestamp":1776845449000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11481063\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":94,"URL":"https:\/\/doi.org\/10.1109\/access.2026.3683761","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}