{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:43:51Z","timestamp":1776883431872,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758312","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:51Z","timestamp":1761377211000},"page":"12564-12570","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["SMIIP-NV: A Multi-Annotation Non-Verbal Expressive Speech Corpus in Mandarin for LLM-Based Speech Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-1238-4829","authenticated-orcid":false,"given":"Zhuojun","family":"Wu","sequence":"first","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China and Duke Kunshan University, Kunshan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9419-1580","authenticated-orcid":false,"given":"Dong","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China and Duke Kunshan University, Kunshan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9344-7415","authenticated-orcid":false,"given":"Juan","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Wuhan University, Wuhan, China and School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7957-9058","authenticated-orcid":false,"given":"Yechen","family":"Wang","sequence":"additional","affiliation":[{"name":"OfSpectrum, Inc., Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9315-0576","authenticated-orcid":false,"given":"Linxi","family":"Li","sequence":"additional","affiliation":[{"name":"OfSpectrum, Inc., Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7825-1229","authenticated-orcid":false,"given":"Liwei","family":"Jin","sequence":"additional","affiliation":[{"name":"OfSpectrum, Inc., Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5883-1195","authenticated-orcid":false,"given":"Hui","family":"Bu","sequence":"additional","affiliation":[{"name":"Beijing AIShell Technology Co. Ltd, China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6838-5160","authenticated-orcid":false,"given":"Pengyuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Speech and Intelligent Information Processing Laboratory, Institute of Acoustics, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6406-1983","authenticated-orcid":false,"given":"Ming","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China and Duke Kunshan University, Kunshan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Better speech synthesis through scaling. arXiv preprint arXiv:2305.07243","author":"Betker James","year":"2023","unstructured":"James Betker. 2023. Better speech synthesis through scaling. arXiv preprint arXiv:2305.07243 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis. arXiv preprint arXiv:2501.04904","author":"Cha Jun-Hyeok","year":"2025","unstructured":"Jun-Hyeok Cha, Seung-Bin Kim, Hyung-Seok Oh, and Seong-Whan Lee. 2025. JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis. arXiv preprint arXiv:2501.04904 (2025)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1177\/1529100619850176"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890943"},{"key":"e_1_3_2_1_5_1","unstructured":"Zhihao Du Yuxuan Wang Qian Chen Xian Shi Xiang Lv Tianyu Zhao Zhifu Gao Yexin Yang Changfeng Gao Hui Wang et al. 2024. Cosyvoice 2: Scalable streaming speech synthesis with large language models. arXiv preprint arXiv:2412.10117 (2024)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178910"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888737"},{"key":"e_1_3_2_1_8_1","volume-title":"Paraformer: Fast and accurate parallel transformer for non-autoregressive end-to-end speech recognition. arXiv preprint arXiv:2206.08317","author":"Gao Zhifu","year":"2022","unstructured":"Zhifu Gao, Shiliang Zhang, Ian McLoughlin, and Zhijie Yan. 2022. Paraformer: Fast and accurate parallel transformer for non-autoregressive end-to-end speech recognition. arXiv preprint arXiv:2206.08317 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Yonghui Wu, et al.","author":"Jia Ye","year":"2018","unstructured":"Ye Jia, Yu Zhang, Ron Weiss, Quan Wang, Jonathan Shen, Fei Ren, Patrick Nguyen, Ruoming Pang, Ignacio Lopez Moreno, Yonghui Wu, et al., 2018. Transfer learning from speaker verification to multispeaker text-to-speech synthesis. Advances in neural information processing systems, Vol. 31 (2018)."},{"key":"e_1_3_2_1_10_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=mvMI3N4AvD","author":"Jiang Ziyue","year":"2024","unstructured":"Ziyue Jiang, Jinglin Liu, Yi Ren, Jinzheng He, Zhenhui Ye, Shengpeng Ji, Qian Yang, Chen Zhang, Pengfei Wei, Chunfeng Wang, Xiang Yin, Zejun MA, and Zhou Zhao. 2024. Mega-TTS 2: Boosting Prompting Mechanisms for Zero-Shot Speech Synthesis. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=mvMI3N4AvD"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00618"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890672"},{"key":"e_1_3_2_1_13_1","first-page":"11521","article-title":"StoryTTS: A Highly Expressive Text-to-Speech Dataset with Rich Textual Expressiveness Annotations. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Liu Sen","year":"2024","unstructured":"Sen Liu, Yiwei Guo, Xie Chen, and Kai Yu. 2024. StoryTTS: A Highly Expressive Text-to-Speech Dataset with Rich Textual Expressiveness Annotations. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 11521-11525.","journal-title":"IEEE"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2018.2813381"},{"key":"e_1_3_2_1_15_1","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li et al. 2023. Neural codec language models are zero-shot text to speech synthesizers. arXiv preprint arXiv:2301.02111 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Spark-tts: An efficient llm-based text-to-speech model with single-stream decoupled speech tokens. arXiv preprint arXiv:2503.01710","author":"Wang Xinsheng","year":"2025","unstructured":"Xinsheng Wang, Mingqi Jiang, Ziyang Ma, Ziyu Zhang, Songxiang Liu, Linqin Li, Zheng Liang, Qixi Zheng, Rui Wang, Xiaoqin Feng, et al., 2025. Spark-tts: An efficient llm-based text-to-speech model with single-stream decoupled speech tokens. arXiv preprint arXiv:2503.01710 (2025)."},{"key":"e_1_3_2_1_17_1","volume-title":"Laugh Now Cry Later: Controlling Time-Varying Emotional States of Flow-Matching-Based Zero-Shot Text-To-Speech. In 2024 IEEE Spoken Language Technology Workshop (SLT). IEEE, 690-697","author":"Wu Haibin","year":"2024","unstructured":"Haibin Wu, Xiaofei Wang, Sefik Emre Eskimez, Manthan Thakker, Daniel Tompkins, Chung-Hsien Tsai, Canrun Li, Zhen Xiao, Sheng Zhao, Jinyu Li, et al., 2024b. Laugh Now Cry Later: Controlling Time-Varying Emotional States of Flow-Matching-Based Zero-Shot Text-To-Speech. In 2024 IEEE Spoken Language Technology Workshop (SLT). IEEE, 690-697."},{"key":"e_1_3_2_1_18_1","volume-title":"MPE-TTS: Customized Emotion Zero-Shot Text-To-Speech Using Multi-Modal Prompt. arXiv preprint arXiv:2505.18453","author":"Wu Zhichao","year":"2025","unstructured":"Zhichao Wu, Yueteng Kang, Songjun Cao, Long Ma, Qiulin Li, and Qun Yang. 2025. MPE-TTS: Customized Emotion Zero-Shot Text-To-Speech Using Multi-Modal Prompt. arXiv preprint arXiv:2505.18453 (2025)."},{"key":"e_1_3_2_1_19_1","volume-title":"Lightweight Language Model for Speech Synthesis: Attempts and Analysis. In 2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP). IEEE, 501-505","author":"Wu Zhuojun","year":"2024","unstructured":"Zhuojun Wu, Dong Liu, and Ming Li. 2024a. Lightweight Language Model for Speech Synthesis: Attempts and Analysis. In 2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP). IEEE, 501-505."},{"key":"e_1_3_2_1_20_1","volume-title":"The ISCSLP 2024 Conversational Voice Clone (CoVoC) Challenge: Tasks, Results and Findings. In 2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP). IEEE, 506-510","author":"Xia Kangxiang","year":"2024","unstructured":"Kangxiang Xia, Dake Guo, Jixun Yao, Liumeng Xue, Hanzhao Li, Shuai Wang, Zhao Guo, Lei Xie, Qingqing Zhang, Lei Luo, et al., 2024. The ISCSLP 2024 Conversational Voice Clone (CoVoC) Challenge: Tasks, Results and Findings. In 2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP). IEEE, 506-510."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3360885"},{"key":"e_1_3_2_1_22_1","volume-title":"Laughter synthesis using pseudo phonetic tokens with a large-scale in-the-wild laughter corpus. arXiv preprint arXiv:2305.12442","author":"Xin Detai","year":"2023","unstructured":"Detai Xin, Shinnosuke Takamichi, Ai Morimatsu, and Hiroshi Saruwatari. 2023. Laughter synthesis using pseudo phonetic tokens with a large-scale in-the-wild laughter corpus. arXiv preprint arXiv:2305.12442 (2023)."},{"key":"e_1_3_2_1_23_1","volume-title":"NSV-TTS: Non-Speech Vocalization Modeling And Transfer In Emotional Text-To-Speech. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5.","author":"Zhang Haitong","year":"2023","unstructured":"Haitong Zhang, Xinyuan Yu, and Yue Lin. 2023. NSV-TTS: Non-Speech Vocalization Modeling And Transfer In Emotional Text-To-Speech. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5."},{"key":"e_1_3_2_1_24_1","volume-title":"Trung Hieu Nguyen, et al","author":"Zhou Kun","year":"2024","unstructured":"Kun Zhou, You Zhang, Shengkui Zhao, Hao Wang, Zexu Pan, Dianwen Ng, Chong Zhang, Chongjia Ni, Yukun Ma, Trung Hieu Nguyen, et al., 2024. Emotional dimension control in language model-based text-to-speech: Spanning a broad spectrum of human emotions. arXiv preprint arXiv:2409.16681 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758312","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:07:42Z","timestamp":1765339662000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758312"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":24,"alternative-id":["10.1145\/3746027.3758312","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758312","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}