{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T06:07:26Z","timestamp":1784441246397,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":100,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,11,13]],"date-time":"2022-11-13T00:00:00Z","timestamp":1668297600000},"content-version":"vor","delay-in-days":366,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CNS-1949650, CNS-1923778, CNS-1705042"],"award-info":[{"award-number":["CNS-1949650, CNS-1923778, CNS-1705042"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,11,12]]},"DOI":"10.1145\/3460120.3484742","type":"proceedings-article","created":{"date-parts":[[2021,11,13]],"date-time":"2021-11-13T12:05:34Z","timestamp":1636805134000},"page":"235-251","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":29,"title":["\"Hello, It's Me\": Deep Learning-based Speech Synthesis Attacks in the Real World"],"prefix":"10.1145","author":[{"given":"Emily","family":"Wenger","sequence":"first","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Max","family":"Bronckers","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christian","family":"Cianfarani","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jenna","family":"Cryan","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Angela","family":"Sha","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haitao","family":"Zheng","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ben Y.","family":"Zhao","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,11,13]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2015. Announcing WeChat VoicePrint. https:\/\/blog.wechat.com\/2015\/05\/21\/voiceprint-the-new-wechat-password\/."},{"key":"e_1_3_2_1_2_1","unstructured":"2019. Personalize Your Alexa Experience with Voice Profiles. https:\/\/developer.amazon.com\/blogs\/alexa\/post\/1ad16e9b-4f52--4e68--9187-ec2e93faae55\/recognize-voices-and-personalize-your-skills."},{"key":"e_1_3_2_1_3_1","unstructured":"2020. Chase VoiceID. https:\/\/www.chase.com\/personal\/voice-biometrics"},{"key":"e_1_3_2_1_4_1","unstructured":"2020. HSBC VoiceID. https:\/\/www.us.hsbc.com\/customer-service\/voice\/"},{"key":"e_1_3_2_1_5_1","unstructured":"2020. Link Your Voice to your Google Assistant device. https:\/\/support.google.com\/assistant\/answer\/9071681"},{"key":"e_1_3_2_1_6_1","unstructured":"2020. Resemblyzer. https:\/\/github.com\/resemble-ai\/Resemblyzer"},{"key":"e_1_3_2_1_7_1","unstructured":"2020. What Are Alexa Voice Profiles? https:\/\/www.amazon.com\/gp\/help\/customer\/display.html?nodeId=GYCXKY2AB2QWZT2X."},{"key":"e_1_3_2_1_8_1","unstructured":"2021. Attack-VC Github Implementation. https:\/\/github.com\/cyhuang-tw\/attack-vc"},{"key":"e_1_3_2_1_9_1","unstructured":"2021. Lyrebird AI. https:\/\/www.descript.com\/lyrebird"},{"key":"e_1_3_2_1_10_1","unstructured":"2021. Microsoft Azure Speaker Recogition. https:\/\/azure.microsoft.com\/en-us\/services\/cognitive-services\/speaker-recognition\/."},{"key":"e_1_3_2_1_11_1","unstructured":"2021. Mozilla TTS. https:\/\/github.com\/mozilla\/TTS"},{"key":"e_1_3_2_1_12_1","unstructured":"2021. Resemble.AI. https:\/\/www.resemble.ai\/"},{"key":"e_1_3_2_1_13_1","unstructured":"2021. TensorflowTTS. https:\/\/github.com\/TensorSpeech\/TensorFlowTTS"},{"key":"e_1_3_2_1_14_1","unstructured":"2021. Voxforge. http:\/\/www.voxforge.org\/"},{"key":"e_1_3_2_1_15_1","unstructured":"2021. WeChat VoicePrint Documentation. https:\/\/help.wechat.com\/cgi-bin\/micromsg-bin\/oshelpcenter?opcode=2&plat=ios&lang=en&id=150819uqYnUR150819YzINVb."},{"key":"e_1_3_2_1_16_1","volume-title":"Proc. of USENIX.","author":"Ahmed Muhammad Ejaz","year":"2020","unstructured":"Muhammad Ejaz Ahmed, Il-Youp Kwak, Jun Ho Huh, Iljoo Kim, Taekkyung Oh, and Hyoungshick Kim. 2020. Void: A fast and light voice liveness detection system. In Proc. of USENIX."},{"key":"e_1_3_2_1_17_1","volume-title":"CVPR Workshops.","author":"AlBadawy Ehab A","year":"2019","unstructured":"Ehab A AlBadawy, Siwei Lyu, and Hany Farid. 2019. Detecting AI-Synthesized Speech Using Bispectral Analysis.. In CVPR Workshops."},{"key":"e_1_3_2_1_18_1","volume-title":"Proc. of BIOSIG.","author":"Alegre Federico","year":"2014","unstructured":"Federico Alegre, Artur Janicki, and Nicholas Evans. 2014. Re-assessing the threat of replay spoofing attacks against automatic speaker verification. In Proc. of BIOSIG."},{"key":"e_1_3_2_1_19_1","volume-title":"Speech synthesis from neural decoding of spoken sentences. Nature","author":"Anumanchipalli Gopala K","year":"2019","unstructured":"Gopala K Anumanchipalli, Josh Chartier, and Edward F Chang. 2019. Speech synthesis from neural decoding of spoken sentences. Nature (2019)."},{"key":"e_1_3_2_1_20_1","volume-title":"Workshop on Very Large Scale Phonetics Research.","author":"Anumanchipalli Gopala Krishna","year":"2011","unstructured":"Gopala Krishna Anumanchipalli, Kishore Prahallad, and Alan W Black. 2011. Festvox: Tools for creation and analyses of large speech corpora. In Workshop on Very Large Scale Phonetics Research."},{"key":"e_1_3_2_1_21_1","volume-title":"Proc. of NeurIPs","author":"Arik Sercan O","year":"2018","unstructured":"Sercan O Arik, Jitong Chen, Kainan Peng, Wei Ping, and Yanqi Zhou. 2018. Neural voice cloning with a few samples. Proc. of NeurIPs (2018)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.2044-8295.2011.02041.x"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Fr\u00e9d\u00e9ric Bimbot Jean-Fran\u00e7ois Bonastre Corinne Fredouille et al. [n.d.]. A tutorial on text-independent speaker verification. EURASIP Journal on Advances in Signal Processing 2004 ([n. d.]).","DOI":"10.1155\/S1110865704310024"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2008.01.002"},{"key":"e_1_3_2_1_25_1","volume-title":"Who is real bob? adversarial attacks on speaker recognition systems. arXiv preprint arXiv:1911.01840","author":"Chen Guangke","year":"2019","unstructured":"Guangke Chen, Sen Chen, Lingling Fan, Xiaoning Du, Zhe Zhao, Fu Song, and Yang Liu. 2019. Who is real bob? adversarial attacks on speaker recognition systems. arXiv preprint arXiv:1911.01840 (2019)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2017.133"},{"key":"e_1_3_2_1_27_1","unstructured":"Jemine Corentin. 2019. Master thesis : Real-Time Voice Cloning. (2019). https:\/\/matheo.uliege.be\/handle\/2268.2\/6801"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495413"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3045"},{"key":"e_1_3_2_1_30_1","volume-title":"Voice and articulation drillbook","author":"Fairbanks Grant","unstructured":"Grant Fairbanks. 1960. Voice and articulation drillbook. Addison-Wesley Educational Publishers."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1981.1163530"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISECS.2010.65"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383558"},{"key":"e_1_3_2_1_34_1","volume-title":"Acoustical and perceptual study of voice disguise by age modification in speaker verification. Speech Communication","author":"Hautam\u00e4ki Rosa Gonz\u00e1lez","year":"2017","unstructured":"Rosa Gonz\u00e1lez Hautam\u00e4ki, Md Sahidullah, Ville Hautam\u00e4ki, and Tomi Kinnunen. 2017. Acoustical and perceptual study of voice disguise by age modification in speaker verification. Speech Communication (2017)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472652"},{"key":"e_1_3_2_1_36_1","volume-title":"Proc. of ICLR","author":"Hsu Wei-Ning","year":"2019","unstructured":"Wei-Ning Hsu, Yu Zhang, Ron J Weiss, Heiga Zen, Yonghui Wu, Yuxuan Wang, Yuan Cao, Ye Jia, Zhifeng Chen, Jonathan Shen, et al. 2019. Hierarchical generative modeling for controllable speech synthesis. Proc. of ICLR (2019)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-5"},{"key":"e_1_3_2_1_38_1","volume-title":"Proc. of IEEE SLT Workshop","author":"Lin Yist Y","year":"2021","unstructured":"Chien-yu Huang, Yist Y Lin, Hung-yi Lee, and Lin-shan Lee. 2021. Defending Your Voice: Adversarial Attack on Voice Conversion. Proc. of IEEE SLT Workshop (2021)."},{"key":"e_1_3_2_1_39_1","volume-title":"An assessment of automatic speaker verification vulnerabilities to replay spoofing attacks. Security and Communication Networks","author":"Janicki Artur","year":"2016","unstructured":"Artur Janicki, Federico Alegre, and Nicholas Evans. 2016. An assessment of automatic speaker verification vulnerabilities to replay spoofing attacks. Security and Communication Networks (2016)."},{"key":"e_1_3_2_1_40_1","unstructured":"Corentin Jemine. 2020. Real Time Voice Cloning. https:\/\/github.com\/CorentinJ\/Real-Time-Voice-Cloning"},{"key":"e_1_3_2_1_41_1","volume-title":"Proc. of NeurIPs","author":"Jia Ye","year":"2018","unstructured":"Ye Jia, Yu Zhang, Ron Weiss, Quan Wang, Jonathan Shen, Fei Ren, Patrick Nguyen, Ruoming Pang, Ignacio Lopez Moreno, Yonghui Wu, et al. 2018. Transfer learning from speaker verification to multispeaker text-to-speech synthesis. Proc. of NeurIPs (2018)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639535"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853879"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Tomi Kinnunen Md Sahidullah H\u00e9ctor Delgado Massimiliano Todisco Nicholas Evans Junichi Yamagishi and Kong Aik Lee. 2017. The ASVspoof 2017 challenge: Assessing the limits of replay spoofing attack detection. (2017).","DOI":"10.21437\/Interspeech.2017-1111"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6288895"},{"key":"e_1_3_2_1_46_1","unstructured":"John Kominek and Alan W Black. 2003. Carnegie Mellon University ARCTIC databases for speech synthesis. Carnegie Mellon University Language Technologies Institute Tech Report Carnegie Mellon University-LTI-03--177 (2003)."},{"key":"e_1_3_2_1_47_1","first-page":"46","article-title":"Evidence for the reproduction of social class in brief speech","volume":"114","author":"Kraus Michael W.","year":"2019","unstructured":"Michael W. Kraus, Brittany Torrez, Jun Won Park, and Fariba Ghayebi. 2019. Evidence for the reproduction of social class in brief speech. Proc. of National Academy of Sciences 114, 46 (Nov. 2019), 22998--23003.","journal-title":"Proc. of National Academy of Sciences"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462693"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053340"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2009.5326292"},{"key":"e_1_3_2_1_51_1","volume-title":"Proc. of ISIMP.","author":"Lau Yee Wah","year":"2004","unstructured":"Yee Wah Lau, Michael Wagner, and Dat Tran. 2004. Vulnerability of speaker verification to voice mimicking. In Proc. of ISIMP."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-360"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3376897.3377856"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682743"},{"key":"e_1_3_2_1_55_1","volume-title":"Manuel Egu\u00eda, Mariano Sigman, and Marcos A Trevisan.","author":"L\u00f3pez Sabrina","year":"2013","unstructured":"Sabrina L\u00f3pez, Pablo Riera, Mar\u00eda Florencia Assaneo, Manuel Egu\u00eda, Mariano Sigman, and Marcos A Trevisan. 2013. Vocal caricatures reveal signatures of speaker identity. Scientific Reports (2013)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1999-313"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24177-7_30"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","unstructured":"John Mullennix and Steven Stern. 2010. Computer Synthesized Speech Technologies: Tools for Aiding Impairment. IGI Global.","DOI":"10.4018\/978-1-61520-725-1"},{"key":"e_1_3_2_1_59_1","volume-title":"Weidi Xie, and Andrew Zisserman.","author":"Nagrani Arsha","year":"2020","unstructured":"Arsha Nagrani, Joon Son Chung, Weidi Xie, and Andrew Zisserman. 2020. Voxceleb: Large-scalespeaker verificationin the wild. Computer Speech & Language (2020)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2019.23206"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_2_1_62_1","volume-title":"Jan Rozhon, and Miroslav Voznak.","author":"Partila Pavol","year":"2020","unstructured":"Pavol Partila, Jaromir Tovarek, Gokhan Hakki Ilk, Jan Rozhon, and Miroslav Voznak. 2020. Deep Learning Serves Voice Cloning: How Vulnerable Are Automatic Speaker Verification Systems to Spoofing Trials? IEEE Communications Magazine (2020)."},{"key":"e_1_3_2_1_63_1","volume-title":"Pernet and Pascal Belin","author":"Cyril","year":"2012","unstructured":"Cyril R. Pernet and Pascal Belin. 2012. The role of pitch and timbre in voice gender categorization. Frontiers in Psychology 3 (Feb 2012)."},{"key":"e_1_3_2_1_64_1","volume-title":"Proc. of ICLR","author":"Ping Wei","year":"2018","unstructured":"Wei Ping, Kainan Peng, Andrew Gibiansky, Sercan O Arik, Ajay Kannan, Sharan Narang, Jonathan Raiman, and John Miller. 2018. DeepVoice 3: Scaling text-to-speech with convolutional sequence learning. Proc. of ICLR (2018)."},{"key":"e_1_3_2_1_65_1","unstructured":"Kaizhi Qian. 2021. AutoVC Github Implementation. https:\/\/github.com\/auspicious3000\/autovc"},{"key":"e_1_3_2_1_66_1","volume-title":"Proc. of ICML.","author":"Qian Kaizhi","year":"2020","unstructured":"Kaizhi Qian, Yang Zhang, Shiyu Chang, Mark Hasegawa-Johnson, and David Cox. 2020. Unsupervised speech decomposition via triple information bottleneck. In Proc. of ICML."},{"key":"e_1_3_2_1_67_1","volume-title":"Proc. of ICML","author":"Qian Kaizhi","year":"2019","unstructured":"Kaizhi Qian, Yang Zhang, Shiyu Chang, Xuesong Yang, and Mark Hasegawa-Johnson. 2019. Autovc: Zero-shot voice style transfer with only autoencoder loss. Proc. of ICML (2019)."},{"key":"e_1_3_2_1_68_1","volume-title":"Proc. of ICMR.","author":"Qin Yao","year":"2019","unstructured":"Yao Qin, Nicholas Carlini, Garrison Cottrell, Ian Goodfellow, and Colin Raffel. 2019. Imperceptible, robust, and targeted adversarial examples for automatic speech recognition. In Proc. of ICMR."},{"key":"e_1_3_2_1_69_1","volume-title":"ConVoice: Real-Time Zero-Shot Voice Style Transfer with Convolutional Network. arXiv preprint arXiv:2005.07815","author":"Rebryk Yurii","year":"2020","unstructured":"Yurii Rebryk and Stanislav Beliaev. 2020. ConVoice: Real-Time Zero-Shot Voice Style Transfer with Convolutional Network. arXiv preprint arXiv:2005.07815 (2020)."},{"key":"e_1_3_2_1_70_1","volume-title":"Speaker verification using adapted Gaussian mixture models. Digital signal processing 10","author":"Reynolds Douglas A","year":"2000","unstructured":"Douglas A Reynolds, Thomas F Quatieri, and Robert B Dunn. 2000. Speaker verification using adapted Gaussian mixture models. Digital signal processing 10 (2000)."},{"key":"e_1_3_2_1_71_1","volume-title":"Automatic speaker verification: A review","author":"Rosenberg Aaron E","year":"1976","unstructured":"Aaron E Rosenberg. 1976. Automatic speaker verification: A review. IEEE (1976)."},{"key":"e_1_3_2_1_72_1","volume-title":"The coding manual for qualitative researchers","author":"Saldana J.","year":"2009","unstructured":"J. Saldana. 2009. The coding manual for qualitative researchers. Sage Publications Limited (2009)."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1098\/rspb.2010.0769"},{"key":"e_1_3_2_1_74_1","volume-title":"Proc. of NeurIPs","author":"Serr\u00e0 Joan","year":"2019","unstructured":"Joan Serr\u00e0, Santiago Pascual, and Carlos Segura. 2019. Blow: a single-scale hyperconditioned flow for non-parallel raw-audio voice conversion. Proc. of NeurIPs (2019)."},{"key":"e_1_3_2_1_75_1","volume-title":"Talker change detection: A comparison of human and machine performance. The Journal of the Acoustical Society of America","author":"Sharma Neeraj Kumar","year":"2019","unstructured":"Neeraj Kumar Sharma, Shobhana Ganesh, Sriram Ganapathy, and Lori L Holt. 2019. Talker change detection: A comparison of human and machine performance. The Journal of the Acoustical Society of America (2019)."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3427228.3427289"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1145\/2660267.2660274"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/PERCOM.2019.8767399"},{"key":"e_1_3_2_1_80_1","unstructured":"Dan Simmons. 2017. BBC Fools HSBC Voice Recognition System. (2017). https:\/\/www.bbc.com\/news\/technology-39965545"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846260"},{"key":"e_1_3_2_1_82_1","volume-title":"Fraudsters Used AI to Mimic CEO's Voice in Unusual Cybercrime Case. Wall Street Journal (August","author":"Stupp Catherine","year":"2019","unstructured":"Catherine Stupp. 2019. Fraudsters Used AI to Mimic CEO's Voice in Unusual Cybercrime Case. Wall Street Journal (August 2019)."},{"key":"e_1_3_2_1_83_1","volume-title":"Proc. of ICLR","author":"Taigman Yaniv","year":"2018","unstructured":"Yaniv Taigman, Lior Wolf, Adam Polyak, and Eliya Nachmani. 2018. Voiceloop: Voice fitting and synthesis via a phonological loop. Proc. of ICLR (2018)."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-011-0100-4"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"e_1_3_2_1_86_1","volume-title":"Rosa Gonz\u00e1lez Hautam\u00e4ki, and Md Sahidullah","author":"Vestman Ville","year":"2020","unstructured":"Ville Vestman, Tomi Kinnunen, Rosa Gonz\u00e1lez Hautam\u00e4ki, and Md Sahidullah. 2020. Voice mimicry attacks assisted by automatic speaker verification. Computer Speech & Language (2020)."},{"key":"e_1_3_2_1_87_1","volume-title":"Google's AI sounds like a human on the phone -- should we be worried. The Verge (May","author":"Vincent James","year":"2018","unstructured":"James Vincent. 2018. Google's AI sounds like a human on the phone -- should we be worried. The Verge (May 2018)."},{"key":"e_1_3_2_1_88_1","volume-title":"Explicit modelling of session variability for speaker verification. Computer Speech & Language","author":"Vogt Robbie","year":"2008","unstructured":"Robbie Vogt and Sridha Sridharan. 2008. Explicit modelling of session variability for speaker verification. Computer Speech & Language (2008)."},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737422"},{"key":"e_1_3_2_1_91_1","volume-title":"DeepSonar: Towards Effective and Robust Detection of AI-Synthesized Fake Voices. arXiv preprint arXiv:2005.13770","author":"Wang Run","year":"2020","unstructured":"Run Wang, Felix Juefei-Xu, Yihao Huang, Qing Guo, Xiaofei Xie, Lei Ma, and Yang Liu. 2020. DeepSonar: Towards Effective and Robust Detection of AI-Synthesized Fake Voices. arXiv preprint arXiv:2005.13770 (2020)."},{"key":"e_1_3_2_1_92_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"e_1_3_2_1_93_1","unstructured":"Robert Weide. 1998. The Carnegie Mellon pronouncing dictionary. http:\/\/www.speech.cs.cmu.edu\/cgi-bin\/cmudict"},{"key":"e_1_3_2_1_94_1","volume-title":"Speech Accent Archive","author":"Weinberger Steven H.","unstructured":"Steven H. Weinberger. 2013. Speech Accent Archive. George Mason University."},{"key":"e_1_3_2_1_95_1","volume-title":"One-shot voice conversion by vector quantization and u-net architecture. arXiv preprint arXiv:2006.04154","author":"Wu Da-Yi","year":"2020","unstructured":"Da-Yi Wu, Yen-Hao Chen, and Hung-Yi Lee. 2020. Vqvc+: One-shot voice conversion by vector quantization and u-net architecture. arXiv preprint arXiv:2006.04154 (2020)."},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-462"},{"key":"e_1_3_2_1_97_1","doi-asserted-by":"publisher","unstructured":"Junichi Yamagishi Christophe Veaux and Kirsten. MacDonald. [n.d.]. CSTR VCTK Corpus: English Multi-speaker Corpus for CSTR Voice Cloning Toolkit. ([n. d.]). https:\/\/doi.org\/10.7488\/ds\/2645","DOI":"10.7488\/ds\/2645"},{"key":"e_1_3_2_1_98_1","doi-asserted-by":"publisher","DOI":"10.1145\/3319535.3354248"},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.heares.2011.06.008"},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"publisher","DOI":"10.1145\/3133956.3133962"}],"event":{"name":"CCS '21: 2021 ACM SIGSAC Conference on Computer and Communications Security","location":"Virtual Event Republic of Korea","acronym":"CCS '21","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2021 ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3460120.3484742","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3460120.3484742","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3460120.3484742","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T20:48:29Z","timestamp":1763498909000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3460120.3484742"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,12]]},"references-count":100,"alternative-id":["10.1145\/3460120.3484742","10.1145\/3460120"],"URL":"https:\/\/doi.org\/10.1145\/3460120.3484742","relation":{},"subject":[],"published":{"date-parts":[[2021,11,12]]},"assertion":[{"value":"2021-11-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}