{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T17:23:38Z","timestamp":1763227418222,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":72,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T00:00:00Z","timestamp":1720396800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,8]]},"DOI":"10.1145\/3640794.3665880","type":"proceedings-article","created":{"date-parts":[[2024,7,7]],"date-time":"2024-07-07T06:24:56Z","timestamp":1720333496000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Toward a Third-Kind Voice for Conversational Agents in an Era of Blurring Boundaries Between Machine and Human Sounds"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-2421-2186","authenticated-orcid":false,"given":"Jeesun","family":"Oh","sequence":"first","affiliation":[{"name":"Industrial Design, KAIST, Korea, Republic of"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2233-7283","authenticated-orcid":false,"given":"Hyeonjeong","family":"Im","sequence":"additional","affiliation":[{"name":"Industrial Design, KAIST, Korea, Republic of"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3793-6801","authenticated-orcid":false,"given":"Sangsu","family":"Lee","sequence":"additional","affiliation":[{"name":"Industrial Design, KAIST, Korea, Republic of"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,7,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Bing Chat has a secret \u2018Celebrity","author":"Abrams Lawrence","year":"2023","unstructured":"Lawrence Abrams. 2023. Bing Chat has a secret \u2018Celebrity\u2019 mode to impersonate celebrities. https:\/\/www.bleepingcomputer.com\/news\/microsoft\/bing-chat-has-a-secret-celebrity-mode-to-impersonate-celebrities\/. Accessed: 15 August 2023."},{"volume-title":"speak slower. https:\/\/www.aboutamazon.com\/news\/devices\/alexa-speak-slower. Accessed","year":"2023","key":"e_1_3_2_1_2_1","unstructured":"Amazon. 2019. Alexa, speak slower. https:\/\/www.aboutamazon.com\/news\/devices\/alexa-speak-slower. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_3_1","volume-title":"Jackson celebrity voice for Alexa gets an update. https:\/\/www.amazon.science\/latest-news\/samuel-l-jackson-celebrity-voice-for-alexa-gets-an-update. Accessed","author":"Samuel","year":"2023","unstructured":"Amazon. 2020. Samuel L. Jackson celebrity voice for Alexa gets an update. https:\/\/www.amazon.science\/latest-news\/samuel-l-jackson-celebrity-voice-for-alexa-gets-an-update. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3310422"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3371382.3378330"},{"key":"e_1_3_2_1_6_1","unstructured":"Baidu Research. 2017. Deep Voice 3: 2000-Speaker Neural Text-to-Speech. http:\/\/research.baidu.com\/Blog\/index-view?id=91. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Alice Baird Emilia Parada-Cabaleiro Simone Hantke Felix Burkhardt Nicholas Cummins and Bj\u00f6rn Schuller. 2018. The Perception and Analysis of the Likeability and Human Likeness of Synthesized Speech. In Interspeech. https:\/\/api.semanticscholar.org\/CorpusID:52191344","DOI":"10.21437\/Interspeech.2018-1093"},{"key":"e_1_3_2_1_8_1","volume-title":"talk like a Legend. https:\/\/blog.google\/products\/assistant\/talk-like-a-legend\/. Accessed","author":"Bronstein Manuel","year":"2023","unstructured":"Manuel Bronstein. 2019. Hey Google, talk like a Legend. https:\/\/blog.google\/products\/assistant\/talk-like-a-legend\/. Accessed: 15 August 2023."},{"volume-title":"Discourse analysis","author":"Brown Gillian","key":"e_1_3_2_1_9_1","unstructured":"Gillian Brown and George Yule. 1983. Discourse analysis. Cambridge university press."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/2936924.2937059"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376789"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3359325"},{"key":"e_1_3_2_1_13_1","volume-title":"Chen and Cade Metz","author":"X.","year":"2019","unstructured":"Brian\u00a0X. Chen and Cade Metz. 2019. Google\u2019s Duplex Uses A.I. to Mimic Humans (Sometimes). https:\/\/www.nytimes.com\/2019\/05\/22\/technology\/personaltech\/ai-google-duplex.html. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376461"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376569"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300705"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1336"},{"key":"e_1_3_2_1_18_1","volume-title":"Mapping 24 emotions conveyed by brief human vocalization.The American psychologist","author":"Cowen S.","year":"2019","unstructured":"Alan\u00a0S. Cowen, Hillary\u00a0Anger Elfenbein, Petri Laukka, and Dacher Keltner. 2019. Mapping 24 emotions conveyed by brief human vocalization.The American psychologist (2019). https:\/\/api.semanticscholar.org\/CorpusID:58563174"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581281"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517564"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3338286.3340116"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308532.3329435"},{"key":"e_1_3_2_1_23_1","volume-title":"WaveNet: A generative model for raw audio. https:\/\/www.deepmind.com\/blog\/wavenet-a-generative-model-for-raw-audio. Accessed","author":"DeepMind Google","year":"2023","unstructured":"Google DeepMind. 2016. WaveNet: A generative model for raw audio. https:\/\/www.deepmind.com\/blog\/wavenet-a-generative-model-for-raw-audio. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","unstructured":"A. Govender and S. King. 2018. Measuring the cognitive load of synthetic speech using a dual task paradigm. In 19th Annual Conference of the International Speech Communication INTERSPEECH 2018. International Speech Communication Association 2843\u20132847. https:\/\/doi.org\/10.21437\/Interspeech.2018-1199 Conference code: 139961.","DOI":"10.21437\/Interspeech.2018-1199"},{"key":"e_1_3_2_1_25_1","unstructured":"Xuedong Huang. 2018. Microsoft\u2019s new neural text-to-speech service helps machines speak like people. https:\/\/azure.microsoft.com\/en-us\/blog\/microsoft-s-new-neural-text-to-speech-service-helps-machines-speak-like-people\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3532106.3533528"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445579"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1177\/002383099804100405"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.3389\/fnbot.2020.593732"},{"key":"e_1_3_2_1_30_1","volume-title":"Google Duplex: An AI System for Accomplishing Real-World Tasks Over the Phone. https:\/\/ai.googleblog.com\/2018\/05\/duplex-ai-system-for-natural-conversation.html. Accessed","author":"Leviathan Yaniv","year":"2018","unstructured":"Yaniv Leviathan. 2018. Google Duplex: An AI System for Accomplishing Real-World Tasks Over the Phone. https:\/\/ai.googleblog.com\/2018\/05\/duplex-ai-system-for-natural-conversation.html. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_31_1","volume-title":"Detecting disfluency in spontaneous speech","author":"Lickley Robin","year":"1994","unstructured":"Robin Lickley. 1994. Detecting disfluency in spontaneous speech. The University of Edinburgh (01 1994). http:\/\/hdl.handle.net\/1842\/21358"},{"key":"e_1_3_2_1_32_1","volume-title":"Duplex shows Google failing at ethical and creative AI design. https:\/\/techcrunch.com\/2018\/05\/10\/duplex-shows-google-failing-at-ethical-and-creative-ai-design\/. Accessed","author":"Lomas Natasha","year":"2023","unstructured":"Natasha Lomas. 2018. Duplex shows Google failing at ethical and creative AI design. https:\/\/techcrunch.com\/2018\/05\/10\/duplex-shows-google-failing-at-ethical-and-creative-ai-design\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_33_1","volume-title":"D-ID\u2019s new web app gives a face and voice to OpenAI\u2019s ChatGPT. https:\/\/techcrunch.com\/2023\/03\/07\/d-ids-new-web-app-gives-a-face-and-voice-to-openais-chatgpt\/. Accessed","author":"Malik Aisha","year":"2023","unstructured":"Aisha Malik. 2023. D-ID\u2019s new web app gives a face and voice to OpenAI\u2019s ChatGPT. https:\/\/techcrunch.com\/2023\/03\/07\/d-ids-new-web-app-gives-a-face-and-voice-to-openais-chatgpt\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_34_1","volume-title":"Textless NLP: Generating expressive speech from raw audio. https:\/\/ai.facebook.com\/blog\/textless-nlp-generating-expressive-speech-from-raw-audio\/. Accessed","author":"Meta AI.","year":"2023","unstructured":"Meta AI. 2021. Textless NLP: Generating expressive speech from raw audio. https:\/\/ai.facebook.com\/blog\/textless-nlp-generating-expressive-speech-from-raw-audio\/. Accessed: 15 August 2023."},{"volume-title":"d.]. Vall-E. https:\/\/www.microsoft.com\/en-us\/research\/project\/vall-e-x\/. Accessed","year":"2023","key":"e_1_3_2_1_35_1","unstructured":"Microsoft. [n. d.]. Vall-E. https:\/\/www.microsoft.com\/en-us\/research\/project\/vall-e-x\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_36_1","volume-title":"1st International workshop on vocal interactivity in-and-between humans, animals and robots.","author":"Moore K","year":"2017","unstructured":"Roger\u00a0K Moore. 2017. Appropriate voices for artefacts: some key insights. In 1st International workshop on vocal interactivity in-and-between humans, animals and robots."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0747-5632(02)00081-X"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/332040.332452"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1111\/0022-4537.00153"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1559-1816.1997.tb00275.x"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/191666.191703"},{"volume-title":"Wired for speech: How voice activates and advances the human-computer relationship","author":"Nass Clifford\u00a0Ivar","key":"e_1_3_2_1_42_1","unstructured":"Clifford\u00a0Ivar Nass and Scott Brave. 2005. Wired for speech: How voice activates and advances the human-computer relationship. MIT press Cambridge."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-012-0171-x"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/1463160.1463235"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1002\/isaf.1443"},{"key":"e_1_3_2_1_46_1","volume-title":"Navigating the Challenges and Opportunities of Synthetic Voices. https:\/\/openai.com\/blog\/navigating-the-challenges-and-opportunities-of-synthetic-voices. Accessed","author":"AI.","year":"2023","unstructured":"OpenAI. 2024. Navigating the Challenges and Opportunities of Synthetic Voices. https:\/\/openai.com\/blog\/navigating-the-challenges-and-opportunities-of-synthetic-voices. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_47_1","volume-title":"Siri gains a new gender-neutral voice option in latest iOS update. https:\/\/techcrunch.com\/2022\/02\/24\/siri-gains-a-new-gender-neutral-voice-option-in-latest-ios-update\/. Accessed","author":"Perez Sarah","year":"2023","unstructured":"Sarah Perez. 2022. Siri gains a new gender-neutral voice option in latest iOS update. https:\/\/techcrunch.com\/2022\/02\/24\/siri-gains-a-new-gender-neutral-voice-option-in-latest-ios-update\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1080\/02699930441000445"},{"key":"e_1_3_2_1_49_1","first-page":"105","article-title":"Natural language processing for industrial applications","volume":"41","author":"Quarteroni Silvia","year":"2018","unstructured":"Silvia Quarteroni. 2018. Natural language processing for industrial applications. Spektrum 41, 2018 (2018), 105.","journal-title":"Spektrum"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1177\/14614448211024142"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2022.787499"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1177\/0956797617713798"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1177\/1754073919898526"},{"key":"e_1_3_2_1_54_1","unstructured":"Eric\u00a0Hal Schwartz. 2022. New Neosapience Tool Synthesizes Any Text into Emotion for Virtual Actor Speeches \u2013 Exclusive. https:\/\/voicebot.ai\/2022\/09\/14\/new-neosapience-tool-synthesizes-any-text-into-emotion-for-virtual-actor-speeches-exclusive\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_55_1","unstructured":"Eric\u00a0Hal Schwartz. 2023. Synthetic Speech Startup ElevenLabs Raises $2M for AI Voices With Context-Relevant Emotion. https:\/\/voicebot.ai\/2023\/01\/23\/synthetic-speech-startup-elevenlabs-raises-2m-for-ai-voices-with-context-relevant-emotion\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3196709.3196772"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.chb.2019.04.001"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.chb.2019.04.001"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-009-0012-8"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1518\/001872099779656680"},{"key":"e_1_3_2_1_61_1","unstructured":"Suno. 2023. Bark. https:\/\/github.com\/suno-ai\/bark."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300833"},{"key":"e_1_3_2_1_63_1","volume-title":"The first AI that can laugh. https:\/\/elevenlabs.io\/blog\/the_first_ai_that_can_laugh\/. Accessed","author":"Team Elevenlabs","year":"2023","unstructured":"Elevenlabs Team. 2022. The first AI that can laugh. https:\/\/elevenlabs.io\/blog\/the_first_ai_that_can_laugh\/. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_64_1","unstructured":"Sherry Turkle. 2017. Why these friendly robots can\u2019t be good friends to our kids. https:\/\/www.washingtonpost.com\/outlook\/why-these-friendly-robots-cant-be-good-friends-to-our-kids\/2017\/12\/07\/bce1eaea-d54f-11e7-b62d-d9345ced896d_story.html. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_65_1","volume-title":"Encounters with kismet and cog: Children respond to relational artifacts. Digital media: Transformations in human communication 120","author":"Turkle Sherry","year":"2006","unstructured":"Sherry Turkle, Cynthia Breazeal, Olivia Dast\u00e9, and Brian Scassellati. 2006. Encounters with kismet and cog: Children respond to relational artifacts. Digital media: Transformations in human communication 120 (2006)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1111\/spc3.12489"},{"key":"e_1_3_2_1_67_1","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li Lei He Sheng Zhao and Furu Wei. 2023. Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers. arxiv:2301.02111\u00a0[cs.CL]"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1111\/poms.13953"},{"key":"e_1_3_2_1_69_1","volume-title":"Expressive Speech Synthesis with Tacotron. https:\/\/ai.googleblog.com\/2018\/03\/expressive-speech-synthesis-with.html. Accessed","author":"Wang Yuxuan","year":"2023","unstructured":"Yuxuan Wang and RJ Skerry-Ryan. 2018. Expressive Speech Synthesis with Tacotron. https:\/\/ai.googleblog.com\/2018\/03\/expressive-speech-synthesis-with.html. Accessed: 15 August 2023."},{"key":"e_1_3_2_1_70_1","volume-title":"Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135","author":"Wang Yuxuan","year":"2017","unstructured":"Yuxuan Wang, RJ Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron\u00a0J Weiss, Navdeep Jaitly, Zongheng Yang, Ying Xiao, Zhifeng Chen, Samy Bengio, 2017. Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135 (2017)."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jesp.2014.01.005"},{"key":"e_1_3_2_1_72_1","volume-title":"Most famous voice actors. https:\/\/speechify.com\/blog\/most-famous-voice-actors\/. Accessed","author":"Weitzman Cliff","year":"2023","unstructured":"Cliff Weitzman. 2022. Most famous voice actors. https:\/\/speechify.com\/blog\/most-famous-voice-actors\/. Accessed: 15 August 2023."}],"event":{"name":"CUI '24: ACM Conversational User Interfaces 2024","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"],"location":"Luxembourg Luxembourg","acronym":"CUI '24"},"container-title":["ACM Conversational User Interfaces 2024"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665880","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640794.3665880","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T18:03:25Z","timestamp":1755885805000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640794.3665880"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,8]]},"references-count":72,"alternative-id":["10.1145\/3640794.3665880","10.1145\/3640794"],"URL":"https:\/\/doi.org\/10.1145\/3640794.3665880","relation":{},"subject":[],"published":{"date-parts":[[2024,7,8]]},"assertion":[{"value":"2024-07-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}