{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:59:13Z","timestamp":1771700353217,"version":"3.50.1"},"reference-count":40,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100004835","name":"Robotics Institute of Zhejiang University","doi-asserted-by":"publisher","award":["K11801"],"award-info":[{"award-number":["K11801"]}],"id":[{"id":"10.13039\/501100004835","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/access.2021.3065736","type":"journal-article","created":{"date-parts":[[2021,3,12]],"date-time":"2021-03-12T20:44:28Z","timestamp":1615581868000},"page":"42762-42770","source":"Crossref","is-referenced-by-count":3,"title":["Enhancing Local Dependencies for Transformer-Based Text-to-Speech via Hybrid Lightweight Convolution"],"prefix":"10.1109","volume":"9","author":[{"given":"Wei","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref38","first-page":"1","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref33","author":"park","year":"2019","journal-title":"G2Pc"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2123"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054106"},{"key":"ref30","author":"ito","year":"2017","journal-title":"The LJ speech dataset"},{"key":"ref37","first-page":"7654","article-title":"ESPnet-TTS: Unified, reproducible, and integratable open source end-to-end text-to-speech toolkit","author":"hayashi","year":"2020","journal-title":"Proc ICASSP-IEEE Int Conf Acoust Speech Signal Process (ICASSP)"},{"key":"ref36","first-page":"3683","article-title":"Fitting new speakers based on a short untranscribed sample","author":"nachmani","year":"2018","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref35","first-page":"1","article-title":"Voiceloop: Voice fitting and synthesis via a phonological loop","author":"taigman","year":"2018","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref34","author":"park","year":"2019","journal-title":"G2pe"},{"key":"ref10","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Neural Inf Process Syst (NeurIPS)"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"ref11","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2018","journal-title":"arXiv 1810 04805"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref13","first-page":"1","article-title":"XLNet: Generalized autoregressive pretraining for language understanding","author":"yang","year":"2019","journal-title":"Proc Neural Inf Process Syst (NeurIPS)"},{"key":"ref14","first-page":"11","article-title":"Improving language understanding by generative pre-training","volume":"1","author":"radford","year":"2018","journal-title":"OpenAIRE blog"},{"key":"ref15","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"radford","year":"2019","journal-title":"OpenAIRE blog"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref17","first-page":"3171","article-title":"FastSpeech: Fast, robust and controllable text to speech","author":"ren","year":"2019","journal-title":"Proc Neural Inf Process Syst (NeurIPS)"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003949"},{"key":"ref19","first-page":"1","article-title":"QANet: Combining local convolution with global self-attention for reading comprehension","author":"yu","year":"2018","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref28","first-page":"933","article-title":"Language modeling with gated convolutional networks","author":"dauphin","year":"2017","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref4","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"arXiv 1609 03499"},{"key":"ref27","article-title":"Layer normalization","author":"ba","year":"2016","journal-title":"arXiv 1607 06450"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref6","first-page":"1","article-title":"Empirical evaluation of gated recurrent neural networks on sequence modeling","author":"chung","year":"2014","journal-title":"Proc NeurIPS Deep Learn Represent Learn Workshop"},{"key":"ref29","year":"2019","journal-title":"Chinese Standard Mandarin Speech Corpus"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref8","first-page":"1","article-title":"Deep voice 3: Scaling text-to-speech with convolutional sequence learning","author":"ping","year":"2018","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref9","first-page":"1243","article-title":"Convolutional sequence to sequence learning","author":"gehring","year":"2017","journal-title":"Proc Int Conf Mach Learn Res"},{"key":"ref1","first-page":"1","article-title":"Char2wav: End-to-end speech synthesis","author":"sotelo","year":"2017","journal-title":"Proc ICLR Workshop"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016489"},{"key":"ref22","first-page":"1","article-title":"Pay less attention with lightweight and dynamic convolutions","author":"wu","year":"2019","journal-title":"Proc Int Conf Learn Represent (ICLR)"},{"key":"ref21","article-title":"MUSE: Parallel multi-scale attention for sequence to sequence learning","author":"zhao","year":"2019","journal-title":"arXiv 1911 09483"},{"key":"ref24","first-page":"1","article-title":"Rectified linear units improve restricted Boltzmann machines","author":"nair","year":"2010","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref23","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref25","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"J Mach Learn Res"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9312710\/09376936.pdf?arnumber=9376936","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,12,17]],"date-time":"2021-12-17T19:57:28Z","timestamp":1639771048000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9376936\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":40,"URL":"https:\/\/doi.org\/10.1109\/access.2021.3065736","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]}}}