{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T07:41:55Z","timestamp":1784706115900,"version":"3.55.0"},"reference-count":41,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001381","name":"National Research Foundation Singapore","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001381","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Large Language Models Funding Initiative"},{"name":"Japan-Singapore Joint Call: Japan Science and Technology Agency","award":["R24I6IR136"],"award-info":[{"award-number":["R24I6IR136"]}]},{"name":"National Research Foundation, Prime Minister&#x2019;s Office, Singapore"},{"name":"Research Excellence and Technological Enterprise (CREATE) Programme"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Signal Process. Lett."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/lsp.2025.3627104","type":"journal-article","created":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T17:15:21Z","timestamp":1761930921000},"page":"4259-4263","source":"Crossref","is-referenced-by-count":3,"title":["Prompt-Unseen-Emotion: Mixed Emotional Speech Synthesis With Prompt-LLM Contextual Knowledge"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1920-5228","authenticated-orcid":false,"given":"Xiaoxue","family":"Gao","sequence":"first","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A&#x002A;STAR), Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2528-273X","authenticated-orcid":false,"given":"Huayun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A&#x002A;STAR), Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0872-5877","authenticated-orcid":false,"given":"Nancy F.","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A&#x002A;STAR), Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094298"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26488"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.12720\/jait.13.5.398-412"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1093\/ietisy\/e90-d.9.1406"},{"key":"ref5","article-title":"Seamless: Multilingual expressive and streaming speech translation","author":"Barrault","year":"2023"},{"key":"ref6","article-title":"VoiceBench: Benchmarking LLM-based voice assistants","author":"Chen","year":"2024"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053732"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2023-17"},{"key":"ref9","article-title":"Emotional end-to-end neural speech synthesizer","author":"Lee","year":"2017"},{"key":"ref10","article-title":"MM-TTS: A unified framework for multimodal, prompt-induced emotional text-to-speech synthesis","author":"Li","year":"2024","journal-title":"CoRR"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095621"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3402088"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2923951"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413907"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832181"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-465"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP49672.2021.9362069"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.931"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3533357"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2025.3582715"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1511\/2001.4.344"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0103940"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1317"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2022.3233324"},{"key":"ref25","first-page":"5180","article-title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2018"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383524"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29833"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.3390\/app13042225"},{"key":"ref29","article-title":"Emotional dimension control in language model-based text-to-speech: Spanning a broad spectrum of human emotions","author":"Zhou","year":"2024"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888737"},{"key":"ref31","article-title":"Llasa: Scaling train-time and inference-time compute for llama-based speech synthesis","author":"Ye","year":"2025"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1037\/10366-003"},{"key":"ref33","volume-title":"Changing Minds: The Go-To Guide to Mental Health for Family and Friends","author":"Cross","year":"2016"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/odyssey.2024-26"},{"key":"ref35","article-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3747327.3764897"},{"key":"ref37","article-title":"Gemini: A family of highly capable multimodal models","author":"Google","year":"2023"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.640"},{"key":"ref39","article-title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens","author":"Du","year":"2024"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2021.11.006"},{"key":"ref41","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2022"}],"container-title":["IEEE Signal Processing Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/97\/10802935\/11223079.pdf?arnumber=11223079","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:52:50Z","timestamp":1763146370000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11223079\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/lsp.2025.3627104","relation":{},"ISSN":["1070-9908","1558-2361"],"issn-type":[{"value":"1070-9908","type":"print"},{"value":"1558-2361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}