{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,23]],"date-time":"2026-01-23T09:42:50Z","timestamp":1769161370833,"version":"3.49.0"},"reference-count":82,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001381","name":"National Research Foundation Singapore","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]},{"name":"AI Singapore Programme"},{"name":"AISG","award":["AISG-GC-2019-002"],"award-info":[{"award-number":["AISG-GC-2019-002"]}]},{"name":"AISG","award":["AISG-100E-2018-006"],"award-info":[{"award-number":["AISG-100E-2018-006"]}]},{"name":"National Robotics Programme","award":["192 25 00054"],"award-info":[{"award-number":["192 25 00054"]}]},{"name":"Advanced Manufacturing and Engineering Programmatic","award":["A1687b0033"],"award-info":[{"award-number":["A1687b0033"]}]},{"name":"Advanced Manufacturing and Engineering Programmatic","award":["A18A2b0046"],"award-info":[{"award-number":["A18A2b0046"]}]},{"name":"National Key Research and Development Project","award":["2018YFE0122900"],"award-info":[{"award-number":["2018YFE0122900"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61773224"],"award-info":[{"award-number":["61773224"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62066033"],"award-info":[{"award-number":["62066033"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Natural Science Foundation of Inner Mongolia","award":["2018MS06006"],"award-info":[{"award-number":["2018MS06006"]}]},{"DOI":"10.13039\/501100002797","name":"Inner Mongolia Autonomous Region","doi-asserted-by":"publisher","award":["CGZH2018125"],"award-info":[{"award-number":["CGZH2018125"]}],"id":[{"id":"10.13039\/501100002797","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Applied Technology Research and Development Foundation of Inner Mongolia Autonomous Region","award":["2019GG372"],"award-info":[{"award-number":["2019GG372"]}]},{"name":"Applied Technology Research and Development Foundation of Inner Mongolia Autonomous Region","award":["2020GG0046"],"award-info":[{"award-number":["2020GG0046"]}]},{"DOI":"10.13039\/501100007040","name":"Singapore University of Technology and Design","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100007040","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Artificial Intelligence for Human Voice Conversion","award":["SRG ISTD 2020 158"],"award-info":[{"award-number":["SRG ISTD 2020 158"]}]},{"name":"SUTD AI"},{"name":"The Understanding and Synthesis of Expressive Speech by AI"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/taslp.2020.3040523","type":"journal-article","created":{"date-parts":[[2020,11,25]],"date-time":"2020-11-25T21:46:11Z","timestamp":1606340771000},"page":"274-285","source":"Crossref","is-referenced-by-count":27,"title":["Exploiting Morphological and Phonological Features to Improve Prosodic Phrasing for Mongolian Speech Synthesis"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4524-7413","authenticated-orcid":false,"given":"Rui","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8078-3305","authenticated-orcid":false,"given":"Berrak","family":"Sisman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4954-8398","authenticated-orcid":false,"given":"Jichen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guanglai","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9158-9401","authenticated-orcid":false,"given":"Haizhou","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","first-page":"2448","article-title":"A lstm approach with sub-word embeddings for mongolian phrase break prediction","author":"liu","year":"0","journal-title":"Proc 27th Int Conf Comput Linguistics"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-36802-9_68"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639250"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1016\/B0-08-044854-2\/02095-2"},{"key":"ref76","first-page":"99","article-title":"Mongolian text-to-speech system based on deep neural network","author":"liu","year":"2017","journal-title":"Proc Nat Conf Man-Mach Speech Commun"},{"key":"ref77","first-page":"265","article-title":"Tensorflow: A system for large-scale machine learning","author":"abadi","year":"0","journal-title":"Proc 12th USENIX Symp Operating Syst Des Implementation"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-97304-3_17"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2020-33"},{"key":"ref75","first-page":"1","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"0","journal-title":"Proc Int Conf on Learning Rep"},{"key":"ref38","doi-asserted-by":"crossref","first-page":"2149","DOI":"10.21437\/Interspeech.2011-563","article-title":"A grammar based approach to style specific phrase prediction","author":"parlikar","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref78","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"hinton","year":"2014","journal-title":"Journal of Machine Learning Research"},{"key":"ref79","article-title":"Adadelta: An adaptive learning rate method","author":"zeiler","year":"2012"},{"key":"ref33","doi-asserted-by":"crossref","first-page":"797","DOI":"10.1109\/TASL.2008.917071","article-title":"Exploiting acoustic and syntactic features for automatic prosody labeling in a maximum entropy framework","volume":"16","author":"bangalore","year":"2008","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"ref32","doi-asserted-by":"crossref","first-page":"729","DOI":"10.21437\/Interspeech.2004-282","article-title":"Chinese prosody phrase break prediction based on maximum entropy model","author":"li","year":"2004","journal-title":"Proc INTERSPEECH"},{"key":"ref31","first-page":"1","article-title":"Articulation degree as a prosodic dimension of expressive speech","author":"beller","year":"0","journal-title":"Proc International Conf on Speech Prosody"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3038524"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1998.0041"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2010.5684835"},{"key":"ref35","first-page":"282","article-title":"Conditional random fields: Probabilistic models for segmenting and labeling sequence data","volume":"80","author":"lafferty","year":"2001","journal-title":"Proc 18th Int Conf Mach Learn"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/MASSP.1986.1165342"},{"key":"ref60","first-page":"3171","article-title":"Fastspeech: Fast, robust and controllable text to speech","author":"ren","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2018.8706263"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682770"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2922537"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1131"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-48"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2910637"},{"key":"ref65","first-page":"189","article-title":"Significance of word-terminal syllables for prediction of phrase breaks in text-to-speech systems for indian languages","author":"vadapalli","year":"0","journal-title":"Proc ISCA Speech Synthesis Workshop"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1706"},{"key":"ref29","first-page":"2962","article-title":"Deep voice 2: Multi-speaker neural text-to-speech","author":"gibiansky","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2582924"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref69","article-title":"Layer normalization","author":"ba","year":"2016"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2387389"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2012.2227740"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054681"},{"key":"ref22","article-title":"Wavenet: A generative model for raw audio","author":"oord","year":"0","journal-title":"Proc ISCA Speech Synthesis Workshop"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2020-35"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639507"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462237"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-986"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1190"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472760"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404780"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462497"},{"key":"ref56","first-page":"4929","article-title":"Deep semantic role labeling with self-attention","author":"tan","year":"0","journal-title":"Proc 32th AAAI Conf Artif Intell"},{"key":"ref55","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_2"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1472"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-419"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"ref11","first-page":"265","article-title":"Design and evaluation of shared prosodic annotation for spontaneous french speech: from expert knowledge to non-expert annotation","author":"lacheret","year":"0","journal-title":"Proc Linguistic Annotation Workshop"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2014"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"1204","DOI":"10.21437\/Interspeech.2008-362","article-title":"A method for automatic and dynamic estimation of discourse genre typology with prosodic features","author":"obin","year":"2008","journal-title":"Proc INTERSPEECH"},{"key":"ref13","doi-asserted-by":"crossref","first-page":"3070","DOI":"10.21437\/Interspeech.2010-764","article-title":"Expectations for discourse genre identification: A prosodic study","author":"obin","year":"2010","journal-title":"Proc INTERSPEECH"},{"key":"ref14","article-title":"Seen and unseen emotional style transfer for voice conversion with a new emotional speech dataset","author":"zhou","year":"2020"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"4006","DOI":"10.21437\/Interspeech.2017-1452","article-title":"Tacotron: Towards end-to-end speech synthesis","author":"wang","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref16","first-page":"5180","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","volume":"80","author":"wang","year":"0","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-1030"},{"key":"ref17","first-page":"4693","article-title":"Towards end-to-end prosody transfer for expressive speech synthesis with tacotron","volume":"80","author":"skerry-ryan","year":"0","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref81","first-page":"173","article-title":"Character-based joint segmentation and POS tagging for Chinese using bidirectional RNN-CRF","author":"shao","year":"0","journal-title":"Proc 8th Int Joint Conf Natural Lang Process"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639682"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683501"},{"key":"ref80","first-page":"309","article-title":"Attending to characters in neural sequence labeling models","author":"rei","year":"0","journal-title":"Proc 26th Int Conf Comput Linguistics"},{"key":"ref4","article-title":"Graphspeech: Syntax-aware graph attention network for neural speech synthesis","author":"liu","year":"2020"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2020.3016564"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1121\/1.402450"},{"key":"ref5","article-title":"Expressive TTS training with frame and style reconstruction loss","author":"liu","year":"2020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2165280"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1060"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2016394"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2013.2251852"},{"key":"ref46","first-page":"41","article-title":"Learning continuous-valued word representations for phrase break prediction","author":"vadapalli","year":"2014","journal-title":"Proc INTERSPEECH"},{"key":"ref45","doi-asserted-by":"crossref","first-page":"2157","DOI":"10.21437\/Interspeech.2011-565","article-title":"Unsupervised continuous-valued word features for phrase-break prediction without a part-of-speech tagger","author":"watts","year":"2011","journal-title":"Proc INTERSPEECH"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-885"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854070"},{"key":"ref42","first-page":"3111","article-title":"Distributed representations of words and phrases and their compositionality","author":"mikolov","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00804"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref43","first-page":"1","article-title":"Efficient estimation of word representations in vector space","author":"mikolov","year":"0","journal-title":"Proc 1st Int Conf Learn Rep"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9289074\/09271923.pdf?arnumber=9271923","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,29]],"date-time":"2022-11-29T21:37:54Z","timestamp":1669757874000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9271923\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":82,"URL":"https:\/\/doi.org\/10.1109\/taslp.2020.3040523","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]}}}