{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T05:58:08Z","timestamp":1785563888448,"version":"3.56.0"},"reference-count":23,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1109\/asru65441.2025.11434642","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T18:08:34Z","timestamp":1785521314000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["StellarTTS: Sparse Temporal Embedding for Low-Latency and Robust Speech Synthesis"],"prefix":"10.1109","author":[{"given":"Kaicheng","family":"Luo","sequence":"first","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuefei","family":"Gong","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yutao","family":"Sun","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinling","family":"He","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujie","family":"Hou","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoyang","family":"Xing","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huiyan","family":"Li","sequence":"additional","affiliation":[{"name":"Honor Device Co., Ltd.,"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bing","family":"Han","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University","place":["Shanghai, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanmin","family":"Qian","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University","place":["Shanghai, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Neural codec language models are zero-shot text to speech synthesizers","author":"Wang","year":"2023","journal-title":"arXiv preprint arXiv:2301.02111"},{"key":"ref2","article-title":"Cosyvoice: A scalable multilingual zeroshot text-to-speech synthesizer based on supervised semantic tokens","author":"Du","year":"2024","journal-title":"arXiv preprint arXiv:2407.05407"},{"key":"ref3","article-title":"Cosyvoice 2: Scalable streaming speech synthesis with large language models","volume-title":"arXiv preprint arXiv:2412.10117","author":"Du","year":"2024"},{"key":"ref4","article-title":"Soundstorm: Efficient parallel audio generation","author":"Borsos","year":"2023","journal-title":"arXiv preprint arXiv:2305.09636"},{"key":"ref5","article-title":"Maskgct: Zero-shot text-tospeech with masked generative codec transformer","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv:2409.00750"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832320"},{"key":"ref7","article-title":"F5TTS: A fairytaler that fakes fluent and faithful speech with flow matching","author":"Chen","year":"2024","journal-title":"arXiv preprint arXiv:2410.06885"},{"key":"ref8","article-title":"Naturalspeech 3: Zeroshot speech synthesis with factorized codec and diffusion models","volume-title":"arXiv preprint arXiv:2403.03100","author":"Ju","year":"2024"},{"key":"ref9","first-page":"3165","article-title":"Fastspeech: Fast, robust and controllable text to speech","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems","author":"Ren"},{"key":"ref10","article-title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","author":"Ren","year":"2020","journal-title":"arXiv preprint arXiv:2006.04558"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.3362\/0262-8104.2002.009"},{"key":"ref12","article-title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers","author":"Shen","year":"2023","journal-title":"arXiv preprint arXiv:2304.09116"},{"key":"ref13","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv preprint arXiv:2302.13971"},{"key":"ref14","article-title":"Seed-tts: A family of high-quality versatile speech generation models","author":"Anastassiou","year":"2024","journal-title":"arXiv preprint arXiv:2406.02430"},{"key":"ref15","first-page":"6309","article-title":"Neural discrete representation learning","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems","author":"Van"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"ref17","article-title":"High fidelity neural audio compression","author":"D\u00e9fossez","year":"2022","journal-title":"arXiv preprint arXiv:2210.13438"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"ref21","article-title":"3D-Speaker: A Large-Scale MultiDevice, Multi-Distance, and Multi-Dialect Corpus for Speech Representation Disentanglement","author":"Zheng","year":"2023","journal-title":"arXiv preprint arXiv:2306.15354"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832365"},{"key":"ref23","article-title":"FireRedTTS: A foundation text-to-speech framework for industry-level generative speech applications","author":"Guo","year":"2024","journal-title":"arXiv preprint arXiv:2409.03283"}],"event":{"name":"2025 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,12,6]]},"end":{"date-parts":[[2025,12,10]]}},"container-title":["2025 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11434577\/11433836\/11434642.pdf?arnumber=11434642","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T05:10:30Z","timestamp":1785561030000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11434642\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12]]},"references-count":23,"URL":"https:\/\/doi.org\/10.1109\/asru65441.2025.11434642","relation":{},"subject":[],"published":{"date-parts":[[2025,12]]}}}