{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T03:38:59Z","timestamp":1771299539700,"version":"3.50.1"},"reference-count":33,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"China State Shipbuilding Corporation (CSSC) Guangxi Shipbuilding and Offshore Engineering Technology Collaboration Project","award":["ZCGXJSB20226300222-06"],"award-info":[{"award-number":["ZCGXJSB20226300222-06"]}]},{"name":"100 Scholar Plan of the Guangxi Zhuang Autonomous Region of China","award":["2018"],"award-info":[{"award-number":["2018"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3283772","type":"journal-article","created":{"date-parts":[[2023,6,7]],"date-time":"2023-06-07T17:38:12Z","timestamp":1686159492000},"page":"57674-57682","source":"Crossref","is-referenced-by-count":12,"title":["MixGAN-TTS: Efficient and Stable Speech Synthesis Based on Diffusion Model"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0778-6144","authenticated-orcid":false,"given":"Yan","family":"Deng","sequence":"first","affiliation":[{"name":"School of Computer, Electronics and Information, Guangxi University, Nanning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4951-6337","authenticated-orcid":false,"given":"Ning","family":"Wu","sequence":"additional","affiliation":[{"name":"Key Laboratory of Beibu Gulf Offshore Engineering Equipment and Technology, Beibu Gulf University, Qinzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2264-8866","authenticated-orcid":false,"given":"Chengjun","family":"Qiu","sequence":"additional","affiliation":[{"name":"College of Mechanical Naval Architecture and Ocean Engineering, Beibu Gulf University, Qinzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3533-3619","authenticated-orcid":false,"given":"Yangyang","family":"Luo","sequence":"additional","affiliation":[{"name":"School of Computer, Electronics and Information, Guangxi University, Nanning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9950-684X","authenticated-orcid":false,"given":"Yan","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer, Electronics and Information, Guangxi University, Nanning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","first-page":"8067","article-title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search","volume":"33","author":"kim","year":"2020","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref12","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","author":"ren","year":"2020","journal-title":"arXiv 2006 04558"},{"key":"ref15","article-title":"DiffGAN-TTS: High-fidelity and efficient text-to-speech with denoising diffusion GANs","author":"liu","year":"2022","journal-title":"arXiv 2201 11972"},{"key":"ref14","first-page":"13963","article-title":"PortaSpeech: Portable and high-quality generative text-to-speech","volume":"34","author":"ren","year":"2021","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.819861"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-19551-8_23"},{"key":"ref11","first-page":"1","article-title":"FastSpeech: Fast, robust and controllable text to speech","volume":"32","author":"ren","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-74048-3_4"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.iswa.2022.200077"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2954342"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21350"},{"key":"ref16","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"ho","year":"2020","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref19","first-page":"8599","article-title":"Grad-TTS: A diffusion probabilistic model for text-to-speech","author":"popov","year":"2021","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref18","article-title":"Wavegrad: Estimating gradients for waveform generation","author":"chen","year":"2020","journal-title":"arXiv 2009 00713"},{"key":"ref24","article-title":"Searching for activation functions","author":"ramachandran","year":"2017","journal-title":"arXiv 1710 05941"},{"key":"ref23","first-page":"1","article-title":"Attention is all you need","volume":"30","author":"vaswani","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-971"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.304"},{"key":"ref20","article-title":"Symbolic music generation with diffusion models","author":"mittal","year":"2021","journal-title":"arXiv 2103 16091"},{"key":"ref22","article-title":"AISHELL-3: A multi-speaker Mandarin TTS corpus and the baselines","author":"shi","year":"2020","journal-title":"arXiv 2010 11567"},{"key":"ref21","article-title":"Tackling the generative learning trilemma with denoising diffusion GANs","author":"xiao","year":"2021","journal-title":"arXiv 2112 07804"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"ref27","first-page":"1558","article-title":"Autoencoding beyond pixels using a learned similarity metric","author":"larsen","year":"2016","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref29","author":"chu","year":"2006","journal-title":"Objective measure for estimating mean opinion score of synthesized speech"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-019-08198-5"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1600"},{"key":"ref6","first-page":"17022","article-title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"kong","year":"2020","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref5","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016","journal-title":"arXiv 1609 03499"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10145456.pdf?arnumber=10145456","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,3]],"date-time":"2023-07-03T18:31:07Z","timestamp":1688409067000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10145456\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3283772","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}