{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:44Z","timestamp":1776931184796,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,13]]},"DOI":"10.1145\/3747327.3764897","type":"proceedings-article","created":{"date-parts":[[2025,10,11]],"date-time":"2025-10-11T14:04:34Z","timestamp":1760191474000},"page":"199-204","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["MultiGen: Child-Friendly Multilingual Speech Generator with LLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1920-5228","authenticated-orcid":false,"given":"Xiaoxue","family":"Gao","sequence":"first","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A*STAR), Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2528-273X","authenticated-orcid":false,"given":"Huayun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A*STAR), Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0872-5877","authenticated-orcid":false,"given":"Nancy","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute for Infocomm Research, Agency for Science, Technology, and Research (A*STAR), Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,12]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30139-4_5"},{"key":"e_1_3_3_2_3_2","unstructured":"Li-Wei Chen Shinji Watanabe and Alexander Rudnicky. 2023. A vector quantized approach for text to speech synthesis on real-world spontaneous speech. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.04215 (2023)."},{"key":"e_1_3_3_2_4_2","unstructured":"Sanyuan Chen Shujie Liu Long Zhou Yanqing Liu Xu Tan Jinyu Li Sheng Zhao Yao Qian and Furu Wei. 2024. VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.05370 (2024)."},{"key":"e_1_3_3_2_5_2","unstructured":"Yiming Chen Xianghu Yue Chen Zhang Xiaoxue Gao Robby\u00a0T Tan and Haizhou Li. 2024. Voicebench: Benchmarking llm-based voice assistants. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.17196 (2024)."},{"key":"e_1_3_3_2_6_2","unstructured":"Zhihao Du Qian Chen Shiliang Zhang Kai Hu Heng Lu Yexin Yang Hangrui Hu Siqi Zheng Yue Gu Ziyang Ma et\u00a0al. 2024. Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.05407 (2024)."},{"key":"e_1_3_3_2_7_2","unstructured":"Xiaoxue Gao Yiming Chen Xianghu Yue Yu Tsao and Nancy\u00a0F Chen. 2025. TTSlow: Slow Down Text-to-Speech with Efficiency Robustness Evaluations. IEEE Transactions on Audio Speech and Language Processing (2025)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC47483.2019.9023056"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2020-36"},{"key":"e_1_3_3_2_10_2","unstructured":"Xiaoxue Gao Chen Zhang Yiming Chen Huayun Zhang and Nancy\u00a0F Chen. 2024. Emo-dpo: Controllable emotional speech synthesis through direct preference optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.10157 (2024)."},{"key":"e_1_3_3_2_11_2","unstructured":"Xiaoxue Gao Huayun Zhang and Nancy\u00a0F Chen. 2025. Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.02742 (2025)."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-969"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Arthur\u00a0C Graesser Mark\u00a0W Conley and Andrew Olney. 2012. Intelligent tutoring systems. (2012).","DOI":"10.1037\/13275-018"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Foteini Grivokostopoulou Isidoros Perikos and Ioannis Hatzilygeroudis. 2017. An educational system for learning search algorithms and automatically assessing student performance. International Journal of Artificial Intelligence in Education 27 1 (2017) 207\u2013240.","DOI":"10.1007\/s40593-016-0116-x"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"crossref","unstructured":"Jason\u00a0M Harley Fran\u00e7ois Bouchet M\u00a0Sazzad Hussain Roger Azevedo and Rafael Calvo. 2015. A multi-componential analysis of emotions during complex learning with an intelligent multi-agent system. Computers in Human Behavior 48 (2015) 615\u2013625.","DOI":"10.1016\/j.chb.2015.02.013"},{"key":"e_1_3_3_2_16_2","unstructured":"Yingxu He Zhuohan Liu Shuo Sun Bin Wang Wenyu Zhang Xunlong Zou Nancy\u00a0F Chen and Ai\u00a0Ti Aw. 2024. MERaLiON-AudioLLM: Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.09818 (2024)."},{"key":"e_1_3_3_2_17_2","unstructured":"Muhammad Huzaifah Geyu Lin Tianchi Liu Hardik\u00a0B Sailor Kye\u00a0Min Tan Tarun\u00a0Kumar Vangani Qiongqiong Wang Jeremy\u00a0HM Wong Nancy\u00a0F Chen and Ai\u00a0Ti Aw. 2024. MERaLiON-SpeechEncoder: Towards a Speech Foundation Model for Singapore and Beyond. CoRR (2024)."},{"key":"e_1_3_3_2_18_2","unstructured":"Muhammad Huzaifah Tianchi Liu Hardik\u00a0B Sailor Kye\u00a0Min Tan Tarun\u00a0K Vangani Qiongqiong Wang Jeremy\u00a0HM Wong Nancy\u00a0F Chen and Ai\u00a0Ti Aw. 2024. Towards a Speech Foundation Model for Singapore and Beyond. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.11538 (2024)."},{"key":"e_1_3_3_2_19_2","first-page":"5530","volume-title":"International Conference on Machine Learning","author":"Kim Jaehyeon","year":"2021","unstructured":"Jaehyeon Kim, Jungil Kong, and Juhee Son. 2021. Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In International Conference on Machine Learning. PMLR, 5530\u20135540."},{"key":"e_1_3_3_2_20_2","unstructured":"Jungil Kong Jaehyeon Kim and Jaekyoung Bae. 2020. Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis. Advances in Neural Information Processing Systems 33 (2020) 17022\u201317033."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"James\u00a0A Kulik and John\u00a0D Fletcher. 2016. Effectiveness of intelligent tutoring systems: a meta-analytic review. Review of educational research 86 1 (2016) 42\u201378.","DOI":"10.3102\/0034654315581420"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-2117"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413889"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383524"},{"key":"e_1_3_3_2_25_2","unstructured":"Xiang Li Zhi-Qi Cheng Jun-Yan He Xiaojiang Peng and Alexander\u00a0G Hauptmann. 2024. Mm-tts: A unified framework for multimodal prompt-induced emotional text-to-speech synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.18398 (2024)."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Chien-Chang Lin Anna\u00a0YQ Huang and Owen\u00a0HT Lu. 2023. Artificial intelligence in intelligent tutoring systems toward sustainable education: a systematic review. Smart Learning Environments 10 1 (2023) 41.","DOI":"10.1186\/s40561-023-00260-y"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29833"},{"key":"e_1_3_3_2_28_2","unstructured":"Zhengyuan Liu Geyu Lin Hui\u00a0Li Tan Huayun Zhang Yanfeng Lu Xiaoxue Gao Stella\u00a0Xin Yin He Sun Hock\u00a0Huan Goh Lung\u00a0Hsiang Wong et\u00a0al. 2025. SingaKids: A Multilingual Multimodal Dialogic Tutor for Language Learning. Proceedings of the 63rd annual meeting of the association for computational linguistics (ACL) (2025)."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.296"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.372"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Christopher\u00a0J MacLellan and Kenneth\u00a0R Koedinger. 2022. Domain-general tutor authoring with apprentice learner models. International Journal of Artificial Intelligence in Education 32 1 (2022) 76\u2013117.","DOI":"10.1007\/s40593-020-00214-2"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"Takashi Nose Junichi Yamagishi Takashi Masuko and Takao Kobayashi. 2007. A style control technique for HMM-based expressive speech synthesis. IEICE TRANSACTIONS on Information and Systems 90 9 (2007) 1406\u20131413.","DOI":"10.1093\/ietisy\/e90-d.9.1406"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Benjamin\u00a0D Nye Arthur\u00a0C Graesser and Xiangen Hu. 2014. AutoTutor and family: A review of 17 years of natural language tutoring. International Journal of Artificial Intelligence in Education 24 (2014) 427\u2013469.","DOI":"10.1007\/s40593-014-0029-5"},{"key":"e_1_3_3_2_34_2","first-page":"78","volume-title":"LLM@ AIED","author":"Nye Benjamin\u00a0D","year":"2023","unstructured":"Benjamin\u00a0D Nye, Dillon Mee, and Mark\u00a0G Core. 2023. Generative Large Language Models for Dialog-Based Tutoring: An Early Consideration of Opportunities and Concerns.. In LLM@ AIED. 78\u201388."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"crossref","unstructured":"Aidan Pine Erica Cooper David Guzm\u00e1n Eric Joanis Anna Kazantseva Ross Krekoski Roland Kuhn Samuel Larkin Patrick Littell Delaney Lothian et\u00a0al. 2025. Speech generation for indigenous language education. Computer Speech & Language 90 (2025) 101723.","DOI":"10.1016\/j.csl.2024.101723"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Silvia Pokriv\u010d\u00e1kov\u00e1. 2019. Preparing teachers for the application of AI-powered technologies in foreign language education. Journal of language and cultural education (2019).","DOI":"10.2478\/jolace-2019-0025"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Hongliang Qiao and Aruna Zhao. 2023. Artificial intelligence-based language learning: illuminating the impact on speaking skills and self-regulation in Chinese EFL context. Frontiers in Psychology 14 (2023) 1255594.","DOI":"10.3389\/fpsyg.2023.1255594"},{"key":"e_1_3_3_2_38_2","unstructured":"Yi Ren Yangjun Ruan Xu Tan Tao Qin Sheng Zhao Zhou Zhao and Tie-Yan Liu. 2019. Fastspeech: Fast robust and controllable text to speech. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053732"},{"key":"e_1_3_3_2_40_2","first-page":"5180","volume-title":"International conference on machine learning","author":"Wang Yuxuan","year":"2018","unstructured":"Yuxuan Wang, Daisy Stanton, Yu Zhang, RJ-Skerry Ryan, Eric Battenberg, Joel Shor, Ying Xiao, Ye Jia, Fei Ren, and Rif\u00a0A Saurous. 2018. Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis. In International conference on machine learning. PMLR, 5180\u20135189."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094298"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"Bowen Zhang Nur Afiqah\u00a0Abdul Latiff Justin Kan Rong Tong Donny Soh Xiaoxiao Miao and Ian McLoughlin. 2025. Automated evaluation of children\u2019s speech fluency for low-resource languages. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.19671 (2025).","DOI":"10.21437\/Interspeech.2025-1358"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Huayun Zhang Ke Shi and Nancy\u00a0F Chen. 2021. Multilingual speech evaluation: case studies on English Malay and Tamil. Proc. Interspeech (2021) 4443\u20134447.","DOI":"10.21437\/Interspeech.2021-1258"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"crossref","unstructured":"Ke Zhang and Ayse\u00a0Begum Aslan. 2021. AI technologies for education: Recent research & future directions. Computers and education: Artificial intelligence 2 (2021) 100025.","DOI":"10.1016\/j.caeai.2021.100025"}],"event":{"name":"ICMI Companion '25: Companion Proceedings of the 27th International Conference on Multimodal Interaction","location":"Canberra Australia","acronym":"ICMI Companion '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Companion Proceedings of the 27th International Conference on Multimodal Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3747327.3764897","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,16]],"date-time":"2025-12-16T21:07:40Z","timestamp":1765919260000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3747327.3764897"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,12]]},"references-count":43,"alternative-id":["10.1145\/3747327.3764897","10.1145\/3747327"],"URL":"https:\/\/doi.org\/10.1145\/3747327.3764897","relation":{},"subject":[],"published":{"date-parts":[[2025,10,12]]},"assertion":[{"value":"2025-10-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}