{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T05:34:45Z","timestamp":1725687285690},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icassp49357.2023.10095919","type":"proceedings-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T17:28:30Z","timestamp":1683307710000},"page":"1-5","source":"Crossref","is-referenced-by-count":0,"title":["Singing Voice Synthesis Based on a Musical Note Position-Aware Attention Mechanism"],"prefix":"10.1109","author":[{"given":"Yukiya","family":"Hono","sequence":"first","affiliation":[{"name":"Nagoya Institute of Technology,Nagoya,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kei","family":"Hashimoto","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology,Nagoya,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yoshihiko","family":"Nankaku","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology,Nagoya,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keiichi","family":"Tokuda","sequence":"additional","affiliation":[{"name":"Nagoya Institute of Technology,Nagoya,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Non-attentive Tacotron: Robust and controllable neural tts synthesis including unsupervised duration modeling","author":"shen","year":"2020","journal-title":"arXiv preprint arXiv 2010 09084"},{"key":"ref12","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","author":"ren","year":"2021","journal-title":"Proc ICLR"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1722"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414718"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.5.825"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3118033"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462020"},{"key":"ref32","article-title":"Neural sequenceto-sequence speech synthesis using a hidden semi-markov model based structured attention mechanism","author":"nankaku","year":"2021","journal-title":"arXiv preprint arXiv 2108 13035"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3104165"},{"key":"ref1","first-page":"211","article-title":"Recent development of the HMM-based singing voice synthesis system&#x2013;Sinsy","author":"oura","year":"2010","journal-title":"Proc ISCA SSW7"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1410"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053944"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1109"},{"key":"ref18","article-title":"HiFiSinger: Towards high-fidelity neural singing voice synthesis","author":"chen","year":"2020","journal-title":"arXiv preprint arXiv 2009"},{"key":"ref24","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2015","journal-title":"Proc ICLR"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3552466.3556534"},{"key":"ref26","first-page":"211","article-title":"Initial investigation of encoder-decoder end-to-end tts using marginalization of monotonic hard alignments","author":"yasuda","year":"2019","journal-title":"Proc ISCA SSW10"},{"key":"ref25","first-page":"1293","article-title":"Robust sequence-to-sequence acoustic modeling with stepwise monotonic attention for neural TTS","author":"he","year":"2019","journal-title":"Proc INTERSPEECH"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP49672.2021.9362104"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1399"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414348"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref27","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref8","first-page":"4004","article-title":"Tacotron: Towards end-to-end speech synthesis","author":"wang","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref7","article-title":"Char2wav: End-to-end speech synthesis","author":"sotelo","year":"2017","journal-title":"Proc ICLR Workshop Track"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683154"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1420"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1563"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053811"}],"event":{"name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2023,6,4]]},"location":"Rhodes Island, Greece","end":{"date-parts":[[2023,6,10]]}},"container-title":["ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10094559\/10094560\/10095919.pdf?arnumber=10095919","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T19:03:08Z","timestamp":1699902188000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10095919\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/icassp49357.2023.10095919","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}