{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:55:54Z","timestamp":1781798154184,"version":"3.54.5"},"reference-count":41,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100014188","name":"National Research Foundation of Korea (NRF) Grant through the Korea Government","doi-asserted-by":"publisher","award":["2021R1A2C1014044"],"award-info":[{"award-number":["2021R1A2C1014044"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/access.2022.3175810","type":"journal-article","created":{"date-parts":[[2022,5,17]],"date-time":"2022-05-17T19:45:30Z","timestamp":1652816730000},"page":"52621-52629","source":"Crossref","is-referenced-by-count":11,"title":["Learning to Maximize Speech Quality Directly Using MOS Prediction for Neural Text-to-Speech"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2192-2680","authenticated-orcid":false,"given":"Yeunju","family":"Choi","sequence":"first","affiliation":[{"name":"School of Electrical Engineering, Korea Advanced Institute of Science and Technology (KAIST), Daejeon, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4321-379X","authenticated-orcid":false,"given":"Youngmoon","family":"Jung","sequence":"additional","affiliation":[{"name":"School of Electrical Engineering, Korea Advanced Institute of Science and Technology (KAIST), Daejeon, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Youngjoo","family":"Suh","sequence":"additional","affiliation":[{"name":"Voice Group, Konan Technology Inc., Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8787-6982","authenticated-orcid":false,"given":"Hoirin","family":"Kim","sequence":"additional","affiliation":[{"name":"School of Electrical Engineering, Korea Advanced Institute of Science and Technology (KAIST), Daejeon, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/13.1.25"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1066"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2111"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2018-28"},{"key":"ref30","first-page":"507","article-title":"Large-margin softmax loss for convolutional neural networks","author":"liu","year":"2016","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.444"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2003.12.001"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/VCC_BC.2020-1"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3139"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.2307\/1412159"},{"key":"ref11","article-title":"Deep long audio inpainting","author":"chang","year":"2019","journal-title":"arXiv 1911 06476"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2143"},{"key":"ref13","first-page":"14910","article-title":"MelGAN: Generative adversarial networks for conditional waveform synthesis","author":"kumar","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3076369"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495701"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCPHOT.2018.8368474"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462593"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2019.2953810"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053512"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947440"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref6","first-page":"3171","article-title":"FastSpeech: Fast, robust and controllable text to speech","author":"ren","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682168"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414400"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414870"},{"key":"ref2","first-page":"3918","article-title":"Parallel WaveNet: Fast high-fidelity speech synthesis","author":"van den oord","year":"2018","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref9","article-title":"A neural algorithm of artistic style","author":"gatys","year":"2015","journal-title":"arXiv 1508 06576"},{"key":"ref1","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"arXiv 1609 03499"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2018.2871419"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461965"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2968738"},{"key":"ref24","article-title":"An ASR guided speech intelligibility measure for TTS model selection","author":"baby","year":"2020","journal-title":"arXiv 2006 01463"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2914149"},{"key":"ref23","first-page":"2613","article-title":"Perceptually optimizing the cost function for unit selection in a TTS system with one single run of MOS evaluation","author":"peng","year":"2002","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2003"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383533"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9668973\/09775804.pdf?arnumber=9775804","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T17:40:07Z","timestamp":1746726007000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9775804\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/access.2022.3175810","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]}}}