{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T18:51:08Z","timestamp":1774551068202,"version":"3.50.1"},"reference-count":57,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100002241","name":"Japan Science and Technology Agency (JST), Core Research for Evolutional Science and Technology (CREST), Japan","doi-asserted-by":"publisher","award":["JPMJCR18A6"],"award-info":[{"award-number":["JPMJCR18A6"]}],"id":[{"id":"10.13039\/501100002241","id-type":"DOI","asserted-by":"publisher"}]},{"name":"VoicePersonae Project","award":["JMPMJCR18A6"],"award-info":[{"award-number":["JMPMJCR18A6"]}]},{"DOI":"10.13039\/501100001700","name":"The Ministry of Education, Culture, Sports, Science and Technology (MEXT) KAKENHI, Japan","doi-asserted-by":"publisher","award":["16H06302"],"award-info":[{"award-number":["16H06302"]}],"id":[{"id":"10.13039\/501100001700","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001700","name":"The Ministry of Education, Culture, Sports, Science and Technology (MEXT) KAKENHI, Japan","doi-asserted-by":"publisher","award":["17H04687"],"award-info":[{"award-number":["17H04687"]}],"id":[{"id":"10.13039\/501100001700","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001700","name":"The Ministry of Education, Culture, Sports, Science and Technology (MEXT) KAKENHI, Japan","doi-asserted-by":"publisher","award":["18H04120"],"award-info":[{"award-number":["18H04120"]}],"id":[{"id":"10.13039\/501100001700","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001700","name":"The Ministry of Education, Culture, Sports, Science and Technology (MEXT) KAKENHI, Japan","doi-asserted-by":"publisher","award":["18H04112"],"award-info":[{"award-number":["18H04112"]}],"id":[{"id":"10.13039\/501100001700","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001700","name":"The Ministry of Education, Culture, Sports, Science and Technology (MEXT) KAKENHI, Japan","doi-asserted-by":"publisher","award":["18KT0051"],"award-info":[{"award-number":["18KT0051"]}],"id":[{"id":"10.13039\/501100001700","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2020]]},"DOI":"10.1109\/access.2020.3011975","type":"journal-article","created":{"date-parts":[[2020,7,27]],"date-time":"2020-07-27T21:23:11Z","timestamp":1595884991000},"page":"138149-138161","source":"Crossref","is-referenced-by-count":6,"title":["Modeling of Rakugo Speech and Its Limitations: Toward Speech Synthesis That Entertains Audiences"],"prefix":"10.1109","volume":"8","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5746-6811","authenticated-orcid":false,"given":"Shuhei","family":"Kato","sequence":"first","affiliation":[{"name":"National Institute of Informatics, The Graduate University for Advanced Sciences (SOKENDAI), Hayama, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2130-747X","authenticated-orcid":false,"given":"Yusuke","family":"Yasuda","sequence":"additional","affiliation":[{"name":"National Institute of Informatics, The Graduate University for Advanced Sciences (SOKENDAI), Hayama, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8246-0606","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"additional","affiliation":[{"name":"National Institute of Informatics, Chiyoda, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2978-2793","authenticated-orcid":false,"given":"Erica","family":"Cooper","sequence":"additional","affiliation":[{"name":"National Institute of Informatics, Chiyoda, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7294-7699","authenticated-orcid":false,"given":"Shinji","family":"Takaki","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology, Nagoya, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2752-3955","authenticated-orcid":false,"given":"Junichi","family":"Yamagishi","sequence":"additional","affiliation":[{"name":"National Institute of Informatics, Chiyoda, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"ref38","author":"wenck","year":"1966","journal-title":"The Phonemics of Japanese Questions and Attempts"},{"key":"ref33","author":"yamamoto","year":"2012","journal-title":"Rakugo No Rirekisho&#x2013;Kataritugarete 400-Nen"},{"key":"ref32","author":"nomura","year":"1994","journal-title":"Rakugo No Gengogaku"},{"key":"ref31","year":"0","journal-title":"This Photo is Transformed From &#x2018;DP3M2471&#x2019; by Akira Kawamura Licensed Under CC BY 2 0"},{"key":"ref30","year":"1981","journal-title":"Shumputei Shotaro"},{"key":"ref37","author":"daniels","year":"1958","journal-title":"The Sound System of Standard Japanese A Tentative Account From the Teaching Point of View With a Discussion of the &#x2018;Accent&#x2019; Theory (Accompanied by a Full Translation of a Japanese Account of the Theory) and of Word-Stress"},{"key":"ref36","author":"chiba","year":"1935","journal-title":"A Study of Accent Research Into its Nature and Scope in the Light of Experimental Phonetics"},{"key":"ref35","author":"mori","year":"1929","journal-title":"The Pronunciation of Japanese"},{"key":"ref34","year":"1974","journal-title":"Yanagiya Sanza"},{"key":"ref28","year":"2017","journal-title":"Yose Apuri&#x2014;Warai Suginami Yose Kara"},{"key":"ref27","year":"2007","journal-title":"Shinosuke Radio Rakugo De Date"},{"key":"ref29","year":"1974","journal-title":"Radio Yose"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref20","year":"1897","journal-title":"Shinjuku 3-Chome Shinjuku Tokyo Japan"},{"key":"ref22","year":"1951","journal-title":"23&#x2013;1 Nishi-Ikebukuro"},{"key":"ref21","year":"1964","journal-title":"43&#x2013;12 Asakusa 1-Chome Taito Tokyo Japan"},{"key":"ref24","year":"2004","journal-title":"Asakusa Ochanoma Yose"},{"key":"ref23","year":"2012","journal-title":"Yose Channel"},{"key":"ref26","year":"1978","journal-title":"Shin-Uchi Kyoen"},{"key":"ref25","year":"2011","journal-title":"Kamigata Raukgo No Kai"},{"key":"ref50","first-page":"1","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2015","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1002\/(SICI)1521-4036(200001)42:1<17::AID-BIMJ17>3.0.CO;2-U"},{"key":"ref55","year":"2005","journal-title":"Recommendation G 191 Software Tools and Audio Coding Standardization"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-314"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461452"},{"key":"ref52","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"arXiv 1609 03499"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053436"},{"key":"ref11","year":"2009","journal-title":"Hanashika Miku No Tokusen Rakugo Manju Kowai Desu (in Japanese)"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1227"},{"key":"ref12","year":"2009","journal-title":"[VOCALOID Rakugo] Kamban No Pin [Metsuki Warui Miku] (in Japanese)"},{"key":"ref13","year":"2012","journal-title":"[Hatsune Miku] VOCALOID Rakugo &#x2018;Nozarashi&#x2019; (in Japanese)"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1053"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682353"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-20"},{"key":"ref19","year":"1857","journal-title":"7&#x2013;12 Ueno 2-Chome Taito Tokyo Japan"},{"key":"ref4","first-page":"5180","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","author":"wang","year":"2018","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref3","first-page":"4006","article-title":"Uncovering latent style factors for expressive speech synthesis","author":"wang","year":"2017","journal-title":"Proc Conf Neural Inf Process Syst (NIPS) Mach Learn Audio Signal Process Workshop"},{"key":"ref6","author":"wood","year":"2018","journal-title":"Varying Speaking Styles With Neural Text-to-Speech Alexa Blogs"},{"key":"ref5","article-title":"Towards end-to-end prosody transfer for expressive speech synthesis with tacotron","author":"skerry-ryan","year":"2018","journal-title":"arXiv 1803 09047"},{"key":"ref8","article-title":"MelNet: A generative model for audio in the frequency domain","author":"vasquez","year":"2019","journal-title":"arXiv 1906 01083"},{"key":"ref7","first-page":"1","article-title":"Hierarchical generative modeling for controllable speech synthesis","author":"hsu","year":"2019","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref49","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"arXiv 1706 03762"},{"key":"ref9","article-title":"Effective use of variational embedding capacity in expressive end-to-end speech synthesis","author":"battenberg","year":"2019","journal-title":"arXiv 1906 03402"},{"key":"ref46","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"Proc Conf Neural Inf Process Syst (NIPS)"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462020"},{"key":"ref48","article-title":"Zoneout: Regularizing RNNs by randomly preserving hidden activations","author":"krueger","year":"2017","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref47","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref42","first-page":"1","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref41","article-title":"A guide to convolution arithmetic for deep learning","author":"dumoulin","year":"2016","journal-title":"ArXiv 1603 07285"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/8948470\/09149598.pdf?arnumber=9149598","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,8]],"date-time":"2022-09-08T20:01:23Z","timestamp":1662667283000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9149598\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"references-count":57,"URL":"https:\/\/doi.org\/10.1109\/access.2020.3011975","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]}}}