{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T21:13:51Z","timestamp":1740172431418,"version":"3.37.3"},"reference-count":58,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"UKRI Centre for Doctoral Training in Natural Language Processing"},{"DOI":"10.13039\/100014013","name":"UK Research and Innovation","doi-asserted-by":"publisher","award":["EP\/S022481\/1"],"award-info":[{"award-number":["EP\/S022481\/1"]}],"id":[{"id":"10.13039\/100014013","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000848","name":"University of Edinburgh","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100000848","id-type":"DOI","asserted-by":"publisher"}]},{"name":"School of Informatics and School of Philosophy"},{"name":"Psychology &amp; Language Sciences and Huawei"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/taslp.2023.3273414","type":"journal-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T17:39:07Z","timestamp":1683308347000},"page":"1940-1952","source":"Crossref","is-referenced-by-count":1,"title":["Improving Seq2Seq TTS Frontends With Transcribed Speech Audio"],"prefix":"10.1109","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8947-4207","authenticated-orcid":false,"given":"Siqi","family":"Sun","sequence":"first","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1450-8270","authenticated-orcid":false,"given":"Korin","family":"Richmond","sequence":"additional","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2445-2605","authenticated-orcid":false,"given":"Hao","family":"Tang","sequence":"additional","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054560"},{"key":"ref57","first-page":"67","article-title":"NMT: Open-source toolkit for neural machine translation","author":"klein","year":"0","journal-title":"Proc Assoc Comput Linguistics Syst Demonstrations"},{"article-title":"A survey on neural speech synthesis","year":"2021","author":"tan","key":"ref12"},{"key":"ref56","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc 3rd Int Conf Learn Representations"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-industry.32"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053390"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1599"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-76336-9_3"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1006\/csla.2001.0184"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053512"},{"article-title":"The LJ speech dataset","year":"2017","author":"ito","key":"ref55"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-40"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"article-title":"RNN approaches to text normalization: A challenge","year":"2016","author":"sproat","key":"ref17"},{"journal-title":"Speech and Language Processing","year":"2009","author":"jurafsky","key":"ref16"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054695"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00349"},{"article-title":"The HTK Book (version 3.5a)","year":"2015","author":"young","key":"ref51"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462682"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-47"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-537"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472618"},{"article-title":"On using monolingual corpora in neural machine translation","year":"2015","author":"g\u00fcl\u00e7ehre","key":"ref47"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-6121"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.iwslt-1.2"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178846"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-354"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1392"},{"volume":"2","year":"2019","author":"park","key":"ref8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1461"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21105\/joss.03958"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682353"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2830"},{"key":"ref6","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","author":"ren","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"article-title":"Non-attentive Tacotron: Robust and controllable neural TTS synthesis including unsupervised duration modeling","year":"2020","author":"shen","key":"ref5"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-538"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1162"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1166"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-5403"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1757"},{"key":"ref31","first-page":"214","article-title":"Deep voice 3: 2000-speaker neural text-to-speech","author":"ping","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2007.01.014"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00439"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101183"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.sigmorphon-1.16"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054696"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846248"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178767"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1954"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1208"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-2024"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-134"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9415113"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746447"},{"key":"ref27","first-page":"6","article-title":"T5G2P: Using text-to-text transfer transformer for grapheme-to-phoneme conversion","author":"?ez\u00e1?kov\u00e1","year":"0","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref29","first-page":"195","article-title":"Deep voice: Real-time neural text-to-speech","author":"ar?k","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9970249\/10119221.pdf?arnumber=10119221","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,6,12]],"date-time":"2023-06-12T18:22:33Z","timestamp":1686594153000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10119221\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":58,"URL":"https:\/\/doi.org\/10.1109\/taslp.2023.3273414","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"type":"print","value":"2329-9290"},{"type":"electronic","value":"2329-9304"}],"subject":[],"published":{"date-parts":[[2023]]}}}