{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T15:26:12Z","timestamp":1787066772415,"version":"3.56.0"},"reference-count":72,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T00:00:00Z","timestamp":1751328000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Affective Comput."],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1109\/taffc.2025.3561267","type":"journal-article","created":{"date-parts":[[2025,4,15]],"date-time":"2025-04-15T13:37:37Z","timestamp":1744724257000},"page":"2365-2380","source":"Crossref","is-referenced-by-count":14,"title":["EmoSphere++: Emotion-Controllable Zero-Shot Text-to-Speech Via Emotion-Adaptive Spherical Vector"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4673-9882","authenticated-orcid":false,"given":"Deok-Hyeon","family":"Cho","sequence":"first","affiliation":[{"name":"Department of Artificial Intelligence, Korea University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7229-8123","authenticated-orcid":false,"given":"Hyung-Seok","family":"Oh","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Korea University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2287-9111","authenticated-orcid":false,"given":"Seung-Bin","family":"Kim","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Korea University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6249-4996","authenticated-orcid":false,"given":"Seong-Whan","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, Korea University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1037\/h0077714"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832181"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-398"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1941"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2024.3412152"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003829"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3175578"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383524"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3145293"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3233324"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445996"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747098"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2064"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-610"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3164181"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053732"},{"key":"ref17","first-page":"1","article-title":"Semi-supervised generative modeling for controllable speech synthesis","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Habib"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-307"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683865"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053255"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10636"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/taffc.2025.3530920"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3395994"},{"key":"ref24","article-title":"Emotional dimension control in language model-based text-to-speech: Spanning a broad spectrum of human emotions","author":"Zhou","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-979"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448291"},{"key":"ref27","first-page":"74213","article-title":"P-flow: A fast and data-efficient zero-shot TTS through speech prompting","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.3362\/0262-8104.2002.009"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832320"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1037\/10366-003"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1037\/0022-3514.67.3.525"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.2466\/pms.1986.63.3.1156"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1080\/02699939208411068"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.876118"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992253"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/0092-6566(77)90037-X"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.cogsys.2018.01.004"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2023.3280038"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2023.3243463"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2019.2945322"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2016.2598741"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3349053"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2018.2828429"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126281"},{"key":"ref45","first-page":"5180","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054556"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3268571"},{"key":"ref48","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Casanova"},{"key":"ref49","first-page":"10970","article-title":"GenerSpeech: Towards style transfer for generalizable out-of-domain text-to-speech","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Huang"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-58347-1_10"},{"key":"ref51","first-page":"6309","article-title":"Neural discrete representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Van Den"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01211"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3263585"},{"issue":"11","key":"ref54","article-title":"A review of statistical outlier methods","volume":"30","author":"Walfish","year":"2006","journal-title":"Pharmaceut. Technol."},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.931"},{"key":"ref56","first-page":"1","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jia"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-50381-8_19"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.3390\/electronics13071380"},{"key":"ref61","first-page":"1","article-title":"Flow matching for generative modeling","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lipman"},{"key":"ref62","first-page":"8599","article-title":"Grad-TTS: A diffusion probabilistic model for text-to-speech","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Popov"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2021.11.006"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2736999"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"ref66","first-page":"365","article-title":"The festival speech synthesis system, version 1.4. 2","volume":"6","author":"Black","year":"2001","journal-title":"Unpublished Document Available Via"},{"key":"ref67","first-page":"1","article-title":"BigVGAN: A universal neural vocoder with large-scale training","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Lee"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"ref69","first-page":"8067","article-title":"Glow-TTS: A generative flow for text-to-speech via monotonic alignment search","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref70","first-page":"12449","article-title":"Wav2Vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Baevski"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-2441"},{"key":"ref72","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"}],"container-title":["IEEE Transactions on Affective Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5165369\/11152495\/10965917.pdf?arnumber=10965917","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T17:46:42Z","timestamp":1757353602000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10965917\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7]]},"references-count":72,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/taffc.2025.3561267","relation":{},"ISSN":["1949-3045","2371-9850"],"issn-type":[{"value":"1949-3045","type":"electronic"},{"value":"2371-9850","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7]]}}}