{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T12:07:27Z","timestamp":1784635647980,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":70,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T00:00:00Z","timestamp":1784592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"name":"UKRI Centre for Doctoral Training in Natural Language Processing","award":["EP\/S022481\/1"],"award-info":[{"award-number":["EP\/S022481\/1"]}]},{"name":"UKRI AI Centre for Doctoral Training in Responsible and Trustworthy in-the-world Natural Language Processing","award":["EP\/Y030656\/1"],"award-info":[{"award-number":["EP\/Y030656\/1"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,21]]},"DOI":"10.1145\/3816046.3816200","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T11:44:18Z","timestamp":1784634258000},"page":"1-17","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["When Text-to-Speech Speaks in Your Voice: A Study on Public Perception"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-4721-9147","authenticated-orcid":false,"given":"Ariadna","family":"Sanchez","sequence":"first","affiliation":[{"name":"Centre for Speech Technology Research, University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1058-8711","authenticated-orcid":false,"given":"Jinzuomu","family":"Zhong","sequence":"additional","affiliation":[{"name":"Centre for Speech Technology Research, University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3235-8370","authenticated-orcid":false,"given":"Artemis","family":"Deligianni","sequence":"additional","affiliation":[{"name":"School of Informatics, University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9871-4475","authenticated-orcid":false,"given":"Alice","family":"Ross","sequence":"additional","affiliation":[{"name":"Centre for Speech Technology Research, University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2694-2843","authenticated-orcid":false,"given":"Simon","family":"King","sequence":"additional","affiliation":[{"name":"Centre for Speech Technology Research, University of Edinburgh, Edinburgh, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,21]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"[n. d.]. BERTopic - Advanced Transformer-Based Topic Modeling \u2014 bertopic.com. https:\/\/bertopic.com\/. [Accessed 30-12-2025]."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2025-15"},{"key":"e_1_3_3_2_4_2","volume-title":"Proceedings of the 12th International Conference on Communities & Technologies (C&T 2025)","author":"Amirkhani Sima","year":"2025","unstructured":"Sima Amirkhani, Gunnar Stevens, Md Shajalal, and Alexander Boden. 2025. Detecting the Undetectable: Human Judgments and the Challenge of Synthetic Voices. In Proceedings of the 12th International Conference on Communities & Technologies (C&T 2025). European Society for Socially Embedded Technologies (EUSSET)."},{"key":"e_1_3_3_2_5_2","volume-title":"Submitted to The Fourteenth International Conference on Learning Representations","year":"2025","unstructured":"Anonymous. 2025. TTSDS2: Resources and Benchmark for Evaluating Human-Quality Text to Speech Systems. In Submitted to The Fourteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=uGai5lYHlV under review."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.5555\/3327546.3327667"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Kurniawati Azizah. 2024. Zero-shot voice cloning text-to-speech for dysphonia disorder speakers. IEEE Access 12 (2024) 63528\u201363547.","DOI":"10.1109\/ACCESS.2024.3396377"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2025-1"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Sarah Barrington Emily\u00a0A Cooper and Hany Farid. 2025. People are poorly equipped to detect AI-powered voice clones. Scientific Reports 15 1 (2025) 11004.","DOI":"10.1038\/s41598-025-94170-3"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.eacl-long.50"},{"key":"e_1_3_3_2_11_2","volume-title":"A Political Consultant Faces Charges and Fines for Biden Deepfake Robocalls","author":"Bond Shannon","year":"2024","unstructured":"Shannon Bond. 2024. A Political Consultant Faces Charges and Fines for Biden Deepfake Robocalls. https:\/\/www.npr.org\/2024\/05\/23\/nx-s1-4977582\/fcc-ai-deepfake-robocall-biden-new-hampshire-political-operative"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","unstructured":"Paola Bonifacci Elisa Colombini Michele Marzocchi Valentina Tobia and Lorenzo Desideri. 2022. Text-to-speech applications to reduce mind wandering in students with dyslexia. Journal of Computer Assisted Learning 38 2 (2022) 440\u2013454. arXiv:https:\/\/onlinelibrary.wiley.com\/doi\/pdf\/10.1111\/jcal.1262410.1111\/jcal.12624","DOI":"10.1111\/jcal.12624"},{"key":"e_1_3_3_2_13_2","volume-title":"Fraudsters Cloned Company Director\u2019s Voice In $35 Million Heist, Police Find","author":"Brewster Thomas","year":"2021","unstructured":"Thomas Brewster. 2021. Fraudsters Cloned Company Director\u2019s Voice In $35 Million Heist, Police Find. https:\/\/www.forbes.com\/sites\/thomasbrewster\/2021\/10\/14\/huge-bank-fraud-uses-deep-fake-voice-tech-to-steal-millions\/?sh=6c8658b75591"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2016"},{"key":"e_1_3_3_2_15_2","first-page":"2709","volume-title":"International conference on machine learning","author":"Casanova Edresson","year":"2022","unstructured":"Edresson Casanova, Julian Weber, Christopher\u00a0D Shulby, Arnaldo\u00a0Candido Junior, Eren G\u00f6lge, and Moacir\u00a0A Ponti. 2022. Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone. In International conference on machine learning. PMLR, 2709\u20132720."},{"key":"e_1_3_3_2_16_2","unstructured":"Sanyuan Chen Shujie Liu Long Zhou Yanqing Liu Xu Tan Jinyu Li Sheng Zhao Yao Qian and Furu Wei. 2024. Vall-e 2: Neural codec language models are human parity zero-shot text to speech synthesizers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.05370 (2024)."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Sanyuan Chen Chengyi Wang Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li Lei He Sheng Zhao and Furu Wei. 2025. Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers. IEEE Transactions on Audio Speech and Language Processing 33 (2025) 705\u2013718. 10.1109\/TASLPRO.2025.3530270","DOI":"10.1109\/TASLPRO.2025.3530270"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.313"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"Erica Cooper Wen-Chin Huang Yu Tsao Hsin-Min Wang Tomoki Toda and Junichi Yamagishi. 2024. A review on subjective and objective evaluation of synthetic speech. Acoustical Science and Technology 45 4 (2024) 161\u2013183.","DOI":"10.1250\/ast.e24.12"},{"key":"e_1_3_3_2_20_2","unstructured":"craigsmith. 2023. S06-13 Larger crowd walla; louder voices; big cocktail party; freesound.org. https:\/\/freesound.org\/s\/675041\/. [Accessed 19-12-2025]."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/1240624.1240859"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Joshua\u00a0R De\u00a0Leeuw Rebecca\u00a0A Gilbert and Bj\u00f6rn Luchterhandt. 2023. jsPsych: Enabling an open-source collaborative ecosystem of behavioral experiments. Journal of Open Source Software 8 85 (2023) 5351.","DOI":"10.21105\/joss.05351"},{"key":"e_1_3_3_2_23_2","unstructured":"Zhihao Du Yuxuan Wang Qian Chen Xian Shi Xiang Lv Tianyu Zhao Zhifu Gao Yexin Yang Changfeng Gao Hui Wang et\u00a0al. 2024. Cosyvoice 2: Scalable streaming speech synthesis with large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.10117 (2024)."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-781"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Brooklyne Gipson. 2025. Divide radicalize and conquer: Algorithmic micro-targeting \u201cNot Like Us\u201d propaganda and fractured resistance. Dialogues on Digital Society 1 3 (2025) 428\u2013432.","DOI":"10.1177\/29768640251381450"},{"key":"e_1_3_3_2_26_2","first-page":"1181","volume-title":"Interspeech","author":"Graetzer Simone","year":"2021","unstructured":"Simone Graetzer, Jon Barker, Trevor\u00a0J Cox, Michael Akeroyd, John\u00a0F Culling, Graham Naylor, Eszter Porter, Rhoddy\u00a0Viveros Munoz, et\u00a0al. 2021. Clarity-2021 Challenges: Machine Learning Challenges for Advancing Hearing Aid Processing.. In Interspeech , Vol.\u00a02. 1181\u20131185."},{"key":"e_1_3_3_2_27_2","unstructured":"Maarten Grootendorst. 2022. BERTopic: Neural topic modeling with a class-based TF-IDF procedure. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2203.05794 (2022)."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.5040\/9781350230767"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832365"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"crossref","unstructured":"Joseph Henrich Steven\u00a0J Heine and Ara Norenzayan. 2010. Most people are not WEIRD. Nature 466 7302 (2010) 29\u201329.","DOI":"10.1038\/466029a"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Alexander Hinneburg Heikki Mannila Samuli Kaislaniemi Terttu Nevalainen and Helena Raumolin-Brunberg. 2007. How to handle small samples: bootstrap and Bayesian methods in the analysis of linguistic change. Literary and linguistic computing 22 2 (2007) 137\u2013150.","DOI":"10.1093\/llc\/fqm006"},{"key":"e_1_3_3_2_32_2","unstructured":"Keith Ito and Linda Johnson. 2017. The LJ Speech Dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10887767"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"Yejin Jeon Solee Im Youngjae Kim and Gary\u00a0Geunbae Lee. 2025. Facilitating Personalized TTS for Dysarthric Speakers Using Knowledge Anchoring and Curriculum Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.10412 (2025).","DOI":"10.21437\/Interspeech.2025-596"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/277044.277240"},{"key":"e_1_3_3_2_36_2","first-page":"5530","volume-title":"International Conference on Machine Learning","author":"Kim Jaehyeon","year":"2021","unstructured":"Jaehyeon Kim, Jungil Kong, and Juhee Son. 2021. Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In International Conference on Machine Learning. PMLR, 5530\u20135540."},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2023-7"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"crossref","unstructured":"Courtney\u00a0A Kurinec and Charles\u00a0A Weaver\u00a0III. 2021. \u201cSounding Black\u201d: Speech stereotypicality activates racial stereotypes and expectations about appearance. Frontiers in psychology 12 (2021) 785283.","DOI":"10.3389\/fpsyg.2021.785283"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"S\u00e9bastien Le\u00a0Maguer Simon King and Naomi Harte. 2024. The limits of the mean opinion score for speech synthesis evaluation. Computer Speech & Language 84 (2024) 101577.","DOI":"10.1016\/j.csl.2023.101577"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2331"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/3742413.3789074"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"Kimberly\u00a0T Mai Sergi Bray Toby Davies and Lewis\u00a0D Griffin. 2023. Warning: Humans cannot reliably detect speech deepfakes. Plos one 18 8 (2023) e0285333.","DOI":"10.1371\/journal.pone.0285333"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Alva Markelius Connor Wright Joahna Kuiper Natalie Delille and Yu-Ting Kuo. 2024. The mechanisms of AI hype and its planetary and social costs. AI and Ethics 4 3 (2024) 727\u2013742.","DOI":"10.1007\/s43681-024-00461-2"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732018"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832178"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"crossref","unstructured":"Masahiro Mori Karl\u00a0F MacDorman and Norri Kageki. 2012. The uncanny valley [from the field]. IEEE Robotics & automation magazine 19 2 (2012) 98\u2013100.","DOI":"10.1109\/MRA.2012.2192811"},{"key":"e_1_3_3_2_47_2","volume-title":"Wired for Speech: How Voice Activates and Advances the Human-Computer Relationship","author":"Nass Clifford","year":"2005","unstructured":"Clifford Nass and Scott Brave. 2005. Wired for Speech: How Voice Activates and Advances the Human-Computer Relationship. The MIT Press."},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"crossref","unstructured":"Mireia Ortega Joan\u00a0C Mora and Ingrid Mora-Plaza. 2022. L2 learners\u2019 self-assessment of comprehensibility and accentedness: Over\/under-estimation effects of rating peers and attention to speech features. Pronunciation in Second Language Learning and Teaching Proceedings 12 1 (2022).","DOI":"10.31274\/psllt.13354"},{"key":"e_1_3_3_2_49_2","unstructured":"Spencer Overton. 2024. Overcoming racial harms to democracy from artificial intelligence. Iowa L. Rev. 110 (2024) 805."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2025-33"},{"key":"e_1_3_3_2_51_2","unstructured":"Yi Ren Yangjun Ruan Xu Tan Tao Qin Sheng Zhao Zhou Zhao and Tie-Yan Liu. 2019. Fastspeech: Fast robust and controllable text to speech. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.21437\/SpeechProsody.2024-225"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095057"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-2679"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-405"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"crossref","unstructured":"Kari Spjeldn\u00e6s and Faltin Karlsen. 2024. How digital devices transform literary reading: The impact of e-books audiobooks and online life on reading habits. New media & society 26 8 (2024) 4808\u20134824.","DOI":"10.1177\/14614448221126168"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-2765"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2016.7552917"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-99-0827-1"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446886"},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"crossref","unstructured":"Mohammad\u00a0Shorif Uddin Mahmudul Hasan Tetsuya Shimamura et\u00a0al. 2024. Audio Watermarking: A Comprehensive Review. International Journal of Advanced Computer Science & Applications 15 5 (2024).","DOI":"10.14569\/IJACSA.2024.01505141"},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"crossref","unstructured":"Yuxuan Wang RJ Skerry-Ryan Daisy Stanton Yonghui Wu Ron\u00a0J Weiss Navdeep Jaitly Zongheng Yang Ying Xiao Zhifeng Chen Samy Bengio et\u00a0al. 2017. Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1703.10135 (2017).","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"e_1_3_3_2_64_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Wang Yuancheng","year":"2025","unstructured":"Yuancheng Wang, Haoyue Zhan, Liwei Liu, Ruihong Zeng, Haotian Guo, Jiachen Zheng, Qiang Zhang, Xueyao Zhang, Shunsi Zhang, and Zhizheng Wu. 2025. MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"crossref","unstructured":"Steven\u00a0H Weinberger and Stephen\u00a0A Kunath. 2011. The Speech Accent Archive: towards a typology of English accents. Language & Computers 73 1 (2011).","DOI":"10.1163\/9789401206884_014"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-689"},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-270"},{"key":"e_1_3_3_2_68_2","unstructured":"Junichi Yamagishi Christophe Veaux and Kirsten MacDonald. 2019. CSTR VCTK Corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92). The Rainbow Passage which the speakers read out can be found in the International Dialects of English Archive:(http:\/\/web.ku.edu\/idea\/readings\/rainbow.htm). (2019)."},{"key":"e_1_3_3_2_69_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Zhang Xueyao","year":"2025","unstructured":"Xueyao Zhang, Xiaohui Zhang, Kainan Peng, Zhenyu Tang, Vimal Manohar, Yingru Liu, Jeff Hwang, Dangna Li, Yuhao Wang, Julian Chan, Yuan Huang, Zhizheng Wu, and Mingbo Ma. 2025. Vevo: Controllable Zero-Shot Voice Imitation with Self-Supervised Disentanglement. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=anQDiQZhDP"},{"key":"e_1_3_3_2_70_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-2283"},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888332"}],"event":{"name":"CUI '26: ACM Conversational User Interfaces 2026","location":"Bremen Germany","acronym":"CUI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 8th ACM Conference on Conversational User Interfaces"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3816046.3816200","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T11:48:38Z","timestamp":1784634518000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3816046.3816200"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,21]]},"references-count":70,"alternative-id":["10.1145\/3816046.3816200","10.1145\/3816046"],"URL":"https:\/\/doi.org\/10.1145\/3816046.3816200","relation":{},"subject":[],"published":{"date-parts":[[2026,7,21]]},"assertion":[{"value":"2026-07-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}