{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:45:10Z","timestamp":1776887110146,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3696409.3700163","type":"proceedings-article","created":{"date-parts":[[2024,12,28]],"date-time":"2024-12-28T09:55:23Z","timestamp":1735379723000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["StyleSpeech: Parameter-efficient Fine Tuning for Pre-trained Controllable Text-to-Speech"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-1359-872X","authenticated-orcid":false,"given":"Haowei","family":"Lou","sequence":"first","affiliation":[{"name":"University of New South Wales, Sydney, NSW, AU"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4425-7388","authenticated-orcid":false,"given":"Hye-Young","family":"Paik","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, AU"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4076-1811","authenticated-orcid":false,"given":"Wen","family":"Hu","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, AU"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4149-839X","authenticated-orcid":false,"given":"Lina","family":"Yao","sequence":"additional","affiliation":[{"name":"CSIRO's Data61, Eveleigh, NSW, Sydney, AU"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,28]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Amazon.com Inc.2023. Amazon Alexa. Product website. https:\/\/developer.amazon.com\/en-US\/alexa Accessed: 2023-04-14."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2001.977175"},{"key":"e_1_3_3_1_4_2","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et\u00a0al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_1_5_2","unstructured":"Databaker. 2020. Chinese Mandarin Female Corpus. https:\/\/en.data-baker.com\/datasets\/freeDatasets\/. Accessed: 2023-04-20."},{"key":"e_1_3_3_1_6_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1810.04805 (2018)."},{"key":"e_1_3_3_1_7_2","unstructured":"Google LLC. 2023. Google Assistant. Product website. https:\/\/assistant.google.com Accessed: 2023-04-14."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Daniel Griffin and Jae Lim. 1984. Signal estimation from modified short-time Fourier transform. IEEE Transactions on acoustics speech and signal processing 32 2 (1984) 236\u2013243.","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"e_1_3_3_1_9_2","unstructured":"Mohammad\u00a0Reza Hasanabadi. 2023. An overview of text-to-speech systems and media applications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.14301 (2023)."},{"key":"e_1_3_3_1_10_2","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.09685 (2021)."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541114"},{"key":"e_1_3_3_1_13_2","first-page":"5210","volume-title":"International Conference on Machine Learning","author":"Qian Kaizhi","year":"2019","unstructured":"Kaizhi Qian, Yang Zhang, Shiyu Chang, Xuesong Yang, and Mark Hasegawa-Johnson. 2019. Autovc: Zero-shot voice style transfer with only autoencoder loss. In International Conference on Machine Learning. PMLR, 5210\u20135219."},{"key":"e_1_3_3_1_14_2","first-page":"28492","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International Conference on Machine Learning. PMLR, 28492\u201328518."},{"key":"e_1_3_3_1_15_2","unstructured":"ReadSpeaker. 2024. Virtual Assistant Persona - ReadSpeaker. https:\/\/www.readspeaker.com\/applications\/virtual-assistant-persona\/ Accessed: 2024-04-18."},{"key":"e_1_3_3_1_16_2","unstructured":"Yi Ren Chenxu Hu Xu Tan Tao Qin Sheng Zhao Zhou Zhao and Tie-Yan Liu. 2020. Fastspeech 2: Fast and high-quality end-to-end text to speech. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2006.04558 (2020)."},{"key":"e_1_3_3_1_17_2","unstructured":"Yi Ren Yangjun Ruan Xu Tan Tao Qin Sheng Zhao Zhou Zhao and Tie-Yan Liu. 2019. Fastspeech: Fast robust and controllable text to speech. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"e_1_3_3_1_19_2","unstructured":"Shreyas Seshadri Tuomo Raitio Dan Castellani and Jiangchuan Li. 2021. Emphasis control for parallel neural TTS. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2110.03012 (2021)."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_3_1_21_2","unstructured":"SOVA. 2024. SOVA: Create Virtual Assistant For Any Purpose. https:\/\/sova.ai\/asr-tts\/ Accessed: 2024-04-18."},{"key":"e_1_3_3_1_22_2","unstructured":"Xu Tan Tao Qin Frank Soong and Tie-Yan Liu. 2021. A survey on neural speech synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.15561 (2021)."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Andreas Triantafyllopoulos Bj\u00f6rn\u00a0W Schuller G\u00f6k\u00e7e \u0130ymen Metin Sezgin Xiangheng He Zijiang Yang Panagiotis Tzirakis Shuo Liu Silvan Mertes Elisabeth Andr\u00e9 et\u00a0al. 2023. An overview of affective speech synthesis and conversion in the deep learning era. Proc. IEEE (2023).","DOI":"10.1109\/JPROC.2023.3250266"},{"key":"e_1_3_3_1_24_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Yuxuan Wang RJ Skerry-Ryan Daisy Stanton Yonghui Wu Ron\u00a0J Weiss Navdeep Jaitly Zongheng Yang Ying Xiao Zhifeng Chen Samy Bengio et\u00a0al. 2017. Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1703.10135 (2017).","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Yihao Yao Tao Liang Rui Feng Keke Shi Junxiao Yu Wei Wang and Jianqing Li. 2024. SR-TTS: a rhyme-based end-to-end speech synthesis system. Frontiers in Neurorobotics 18 (2024) 1322312.","DOI":"10.3389\/fnbot.2024.1322312"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1999-596"}],"event":{"name":"MMAsia '24: ACM Multimedia Asia","location":"Auckland New Zealand","acronym":"MMAsia '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 6th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700163","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696409.3700163","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:11Z","timestamp":1750295411000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700163"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":26,"alternative-id":["10.1145\/3696409.3700163","10.1145\/3696409"],"URL":"https:\/\/doi.org\/10.1145\/3696409.3700163","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}