{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T01:05:57Z","timestamp":1740099957928,"version":"3.37.3"},"reference-count":33,"publisher":"IEEE","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,19]]},"DOI":"10.1109\/slt48900.2021.9383562","type":"proceedings-article","created":{"date-parts":[[2021,3,25]],"date-time":"2021-03-25T20:46:54Z","timestamp":1616705214000},"page":"499-506","source":"Crossref","is-referenced-by-count":2,"title":["Lightspeech: Lightweight Non-Autoregressive Multi-Speaker Text-To-Speech"],"prefix":"10.1109","author":[{"given":"Song","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Beibei","family":"Ouyang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingyang","family":"Hong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref30","first-page":"949","article-title":"Advances in joint ctc-attention based end-to-end speech recognition with a deep cnn encoder and rnnlm","author":"hori","year":"2017","journal-title":"Proceedings of the Annual Conference of the International Speech Communication Association INTERSPEECH"},{"key":"ref10","first-page":"2962","article-title":"Deep Voice 2: Multi-speaker neural text-to-speech","author":"gibiansky","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref11","first-page":"214","article-title":"Deep Voice 3: 2000-speaker neural text-to-speech","author":"ping","year":"2018","journal-title":"Proc ICLR"},{"article-title":"DurIANdurian: Duration informed attention network for multimodal synthesis","year":"2019","author":"yu","key":"ref12"},{"article-title":"Parallel neural text-to-speech","year":"2019","author":"peng","key":"ref13"},{"key":"ref14","first-page":"3171","article-title":"FastSpeech: Fast, robust and controllable text to speech","author":"ren","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref15","first-page":"1251","article-title":"Xception: Deep learning with depth-wise separable convolutions","author":"chollet","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"article-title":"Pay less attention with lightweight and dynamic convolutions","year":"2019","author":"wu","key":"ref16"},{"key":"ref17","article-title":"Method, apparatus and computer program providing a multi-speaker database for concatenative text-to-speech synthesis","author":"aaron","year":"2010","journal-title":"US Patent"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-7"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2872060"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854321"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1139"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-1452","article-title":"Tacotron: Towards end-to-end speech synthesis","author":"wang","year":"2017"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/s12046-011-0048-y"},{"article-title":"Close to human quality TTS with transformer","year":"2018","author":"li","key":"ref8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.861820"},{"article-title":"Deep Voice: Real-time neural text-to-speech","year":"2017","author":"arik","key":"ref9"},{"key":"ref1","article-title":"Emotional speech synthesis: A review","author":"schr\u00f6der","year":"2001","journal-title":"Seventh European Conference on Speech Communication and Technology"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054535"},{"key":"ref22","first-page":"10019","article-title":"Neural voice cloning with a few samples","author":"arik","year":"2018","journal-title":"NIPS 2018 The 32nd Annual Conference on Neural Information Processing Systems"},{"key":"ref21","first-page":"4480","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","author":"jia","year":"2018","journal-title":"NIPS 2018 The 32nd Annual Conference on Neural Information Processing Systems"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462020"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054556"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/29.21701"}],"event":{"name":"2021 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2021,1,19]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,1,22]]}},"container-title":["2021 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9383468\/9383452\/09383562.pdf?arnumber=9383562","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,6,8]],"date-time":"2021-06-08T18:52:15Z","timestamp":1623178335000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9383562\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,19]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/slt48900.2021.9383562","relation":{},"subject":[],"published":{"date-parts":[[2021,1,19]]}}}