{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T19:27:02Z","timestamp":1754162822079,"version":"3.41.2"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icassp.2019.8683827","type":"proceedings-article","created":{"date-parts":[[2019,4,17]],"date-time":"2019-04-17T16:01:56Z","timestamp":1555516916000},"page":"5896-5900","source":"Crossref","is-referenced-by-count":12,"title":["Phonemic-level Duration Control Using Attention Alignment for Natural Speech Synthesis"],"prefix":"10.1109","author":[{"given":"Jungbae","family":"Park","sequence":"first","affiliation":[{"name":"Humelo Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kijong","family":"Han","sequence":"additional","affiliation":[{"name":"Korea Advanced Institute of Science and Technology (KAIST)"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuneui","family":"Jeong","sequence":"additional","affiliation":[{"name":"Humelo Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sang Wan","family":"Lee","sequence":"additional","affiliation":[{"name":"Humelo Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/TASLP.2017.2761547"},{"key":"ref11","article-title":"Deep voice: Real-time neural text-to-speech","author":"arik","year":"2017","journal-title":"CoRR"},{"key":"ref12","article-title":"Deep voice 2: Multi-speaker neural text-to-speech","author":"arik","year":"2017","journal-title":"CoRR"},{"key":"ref13","article-title":"Towards end-to-end prosody transfer for expressive speech synthesis with tacotron","author":"skerry-ryan","year":"2018","journal-title":"CoRR"},{"key":"ref14","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","author":"wang","year":"2018","journal-title":"CoRR"},{"key":"ref15","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","volume":"37","author":"xu","year":"2015","journal-title":"Proceedings of The 32nd International Conference on Machine Learning"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1109\/ICASSP.2016.7472621"},{"year":"2017","author":"sotelo","article-title":"Char2wav: End-to-end speech synthesis","key":"ref17"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/ICASSP.2016.7472753"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.21437\/Interspeech.2017-666"},{"year":"2017","author":"ito","article-title":"The lj speech dataset","key":"ref28"},{"key":"ref4","first-page":"1703.10135","article-title":"Tacotron: Towards end-to-end speech synthesis","author":"yuxuan","year":"2017"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/TASSP.1984.1164317"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.18653\/v1\/P16-1101"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.18653\/v1\/D15-1166"},{"year":"2014","author":"bahdanau","article-title":"Neural machine translation by jointly learning to align and translate","key":"ref5"},{"key":"ref8","doi-asserted-by":"crossref","DOI":"10.1109\/ICASSP.2013.6639215","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"zen","year":"2013","journal-title":"2013 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref7","first-page":"1748","article-title":"Sppas: a tool for the phonetic segmentations of speech","author":"bigi","year":"2012","journal-title":"Eighth International Conference on Language Resources and Evaluation"},{"key":"ref2","article-title":"Deep speech: Scaling up end-to-end speech recognition","author":"hannun","year":"2014","journal-title":"arXiv preprint arXiv 1412 5567"},{"key":"ref9","first-page":"2243","article-title":"First step towards end-to-end parametric tts synthesis: Generating spectral parameters with neural attention","author":"xu","year":"2016","journal-title":"Proc Inter-speech"},{"key":"ref1","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"2014","journal-title":"Advances in neural information processing systems"},{"key":"ref20","article-title":"Deep voice 3: Scaling text-to-speech with convolutional sequence learning","author":"ping","year":"2018","journal-title":"Proceedings of International Conference on Learning Representations"},{"key":"ref22","first-page":"3918","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","volume":"80","author":"oord","year":"2018","journal-title":"Proc 35th Int Conf Mach Learn"},{"year":"2016","author":"oord","article-title":"Wavenet: A generative model for raw audio","key":"ref21"},{"key":"ref24","first-page":"1807.0728","article-title":"Clarinet: Parallel wave generation in end-to-end text-to-speech","author":"ping","year":"2018"},{"key":"ref23","first-page":"1712","article-title":"Natural tts synthesis by conditioning wavenet on mel spectrogram predictions","author":"shen","year":"2017"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1109\/SLT.2018.8639672"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/ICASSP.2018.8461829"}],"event":{"name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2019,5,12]]},"location":"Brighton, UK","end":{"date-parts":[[2019,5,17]]}},"container-title":["ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8671773\/8682151\/08683827.pdf?arnumber=8683827","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,30]],"date-time":"2025-07-30T18:40:13Z","timestamp":1753900813000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8683827\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/icassp.2019.8683827","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}