{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T22:03:47Z","timestamp":1780351427174,"version":"3.54.1"},"reference-count":47,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"JSPS KAKENHI","award":["JP18K11163"],"award-info":[{"award-number":["JP18K11163"]}]},{"DOI":"10.13039\/501100005936","name":"CASIO SCIENCE PROMOTION FOUNDATION","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005936","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/taslp.2021.3104165","type":"journal-article","created":{"date-parts":[[2021,8,11]],"date-time":"2021-08-11T20:19:41Z","timestamp":1628713181000},"page":"2803-2815","source":"Crossref","is-referenced-by-count":35,"title":["Sinsy: A Deep Neural Network-Based Singing Voice Synthesis System"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1245-8791","authenticated-orcid":false,"given":"Yukiya","family":"Hono","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kei","family":"Hashimoto","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keiichiro","family":"Oura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yoshihiko","family":"Nankaku","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref38","doi-asserted-by":"crossref","DOI":"10.21437\/Blizzard.2016-11","article-title":"The nitech text-to-speech system for the blizzard challenge 2016","author":"sawada","year":"2016","journal-title":"Proc Blizzard Challenge 2016 Workshop"},{"key":"ref33","first-page":"2021","article-title":"MusicXML definition","year":"0"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.825"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403249"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref37","first-page":"1706","article-title":"An automatic singing skill evaluation method for unknown melodies using pitch interval accuracy and vibrato features","author":"nakano","year":"2006","journal-title":"Proc INTERSPEECH"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853599"},{"key":"ref35","article-title":"Mixture density networks","author":"bishop","year":"1994"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1575"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1999.758104"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053811"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.3390\/app7121313"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1563"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472749"},{"key":"ref15","first-page":"2672","article-title":"Generative adversarial nets","author":"goodfellow","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683154"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2010-188"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1722"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472733"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref27","first-page":"964","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref3","first-page":"211","article-title":"Recent development of the HMM-based singing voice synthesis system-sinsy","author":"oura","year":"2010","journal-title":"Proc ISCA SSW7"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854318"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953087"},{"key":"ref5","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1027"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2047683"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-872"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.23919\/APSIPA.2018.8659797"},{"key":"ref1","first-page":"4009","article-title":"VOCALOID-commercial singing synthesizer based on sample concatenation","author":"kenmochi","year":"2007","journal-title":"Proc INTERSPEECH"},{"key":"ref46","first-page":"1","article-title":"Rap-style singing voice synthesis","volume":"2012 mus 94","author":"saino","year":"0"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1399"},{"key":"ref45","first-page":"2021","article-title":"Sinsy demo.","author":"hono","year":"0"},{"key":"ref22","first-page":"1","article-title":"ByteSing: A chinese singing voice synthesis system using duration allocated encoder-decoder acoustic models and wavernn vocoders","author":"gu","year":"2021","journal-title":"Proc Int Symp Chinese Spoken Language Process"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016-18"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053944"},{"key":"ref42","first-page":"1141","article-title":"An HMM-based singing voice synthesis system","author":"saino","year":"2006","journal-title":"Proc INTERSPEECH"},{"key":"ref24","article-title":"HiFiSinger: Towards high-fidelity neural singing voice synthesis","author":"chen","year":"2020"},{"key":"ref41","first-page":"99","article-title":"Acoustic modeling based on the MDL principle for speech recognition","author":"shinoda","year":"1997","journal-title":"Proc 5th Eur Conf Speech Commun Technol"},{"key":"ref23","first-page":"1306","article-title":"XiaoiceSing: A high-quality and integrated singing voice synthesis system","author":"lu","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref44","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref26","first-page":"6049","article-title":"PeriodNet: A non-autoregressive waveform generation model with a structure separating periodic and aperiodic components","author":"hono","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref43","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc ICLR"},{"key":"ref25","first-page":"76","article-title":"Sequence-to-sequence singing voice synthesis with perceptual entropy loss","author":"shi","year":"2021","journal-title":"Proc ICASSP"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9289074\/09511835.pdf?arnumber=9511835","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T08:33:01Z","timestamp":1725611581000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9511835\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/taslp.2021.3104165","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]}}}