{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T18:41:11Z","timestamp":1770748871306,"version":"3.50.0"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFB3102100"],"award-info":[{"award-number":["2022YFB3102100"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U244120033, U24A20336, 62172243, 62402425, 62402418"],"award-info":[{"award-number":["U244120033, U24A20336, 62172243, 62402425, 62402418"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2024M762829"],"award-info":[{"award-number":["2024M762829"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Provincial Natural Science Foundation","award":["LD24F020002"],"award-info":[{"award-number":["LD24F020002"]}]},{"name":"The Pioneer and Leading Goose R&D Program of Zhejiang","award":["2025C01082, 2025C02033, 2025C02263"],"award-info":[{"award-number":["2025C01082, 2025C02033, 2025C02263"]}]},{"name":"Zhejiang Provincial Priority-Funded Postdoctoral Research Project","award":["ZJ2024001"],"award-info":[{"award-number":["ZJ2024001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755629","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"11638-11647","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Enkidu: Universal Frequential Perturbation for Real-Time Audio Privacy Protection against Voice Deepfakes"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-5301-7019","authenticated-orcid":false,"given":"Zhou","family":"Feng","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5894-662X","authenticated-orcid":false,"given":"Jiahao","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0081-0946","authenticated-orcid":false,"given":"Chunyi","family":"Zhou","sequence":"additional","affiliation":[{"name":"College of Comptuter Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2311-4943","authenticated-orcid":false,"given":"Yuwen","family":"Pu","sequence":"additional","affiliation":[{"name":"School of Big Data &amp; Software Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6085-7300","authenticated-orcid":false,"given":"Qingming","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0896-0690","authenticated-orcid":false,"given":"Tianyu","family":"Du","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4268-372X","authenticated-orcid":false,"given":"Shouling","family":"Ji","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"2685","volume-title":"Proceedings of the 29th USENIX Security Symposium (USENIX Security","author":"Ahmed Muhammad Ejaz","year":"2020","unstructured":"Muhammad Ejaz Ahmed, Il-Youp Kwak, Jun Ho Huh, Iljoo Kim, Taekkyung Oh, and Hyoungshick Kim. 2020. Void: A Fast and Light Voice Liveness Detection System. In Proceedings of the 29th USENIX Security Symposium (USENIX Security 2020). USENIX Association, 2685-2702. https:\/\/www.usenix.org\/conference\/usenixsecurity20\/presentation\/ahmed-muhammad"},{"key":"e_1_3_2_1_2_1","volume-title":"Retrieved","author":"Alspach Kyle","year":"2025","unstructured":"Kyle Alspach. 2024. Audio Deepfake Attacks: Widespread and 'Only Going To Get Worse'. https:\/\/www.crn.com\/news\/ai\/2024\/audio-deepfake-attacks-widespread-and-only-going-to-get-worse, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3537674.3554742"},{"key":"e_1_3_2_1_4_1","volume-title":"Advances in Neural Information Processing Systems 33 (NeurIPS","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. In Advances in Neural Information Processing Systems 33 (NeurIPS 2020). Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/92d1e1eb1cd6f9fba3227870bb6d7f07-Abstract.html"},{"key":"e_1_3_2_1_5_1","first-page":"2691","volume-title":"Proceedings of the 31st USENIX Security Symposium (USENIX Security","author":"Blue Logan","year":"2022","unstructured":"Logan Blue, Kevin Warren, Hadi Abdullah, Cassidy Gibson, Luis Vargas, Jessica O'Dell, Kevin Butler, and Patrick Traynor. 2022. Who Are You (I Really Wanna Know)? Detecting Audio DeepFakes Through Vocal Tract Reconstruction. In Proceedings of the 31st USENIX Security Symposium (USENIX Security 2022). USENIX Association, 2691-2708. https:\/\/www.usenix.org\/conference\/usenixsecurity22\/presentation\/blue"},{"key":"e_1_3_2_1_6_1","first-page":"2709","volume-title":"Proceedings of the 39th International Conference on Machine Learning (ICML","author":"Casanova Edresson","year":"2022","unstructured":"Edresson Casanova, Julian Weber, Christopher Dane Shulby, Arnaldo C\u00e2ndido J\u00fanior, Eren G\u00f6lge, and Moacir A. Ponti. 2022. YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for Everyone. In Proceedings of the 39th International Conference on Machine Learning (ICML 2022). PMLR, 2709-2720. https:\/\/proceedings.mlr.press\/v162\/casanova22a.html"},{"key":"e_1_3_2_1_7_1","unstructured":"Sanyuan Chen Shujie Liu Long Zhou Yanqing Liu Xu Tan Jinyu Li Sheng Zhao Yao Qian and Furu Wei. 2024. VALL-E 2: Neural Codec Language Models are Human Parity Zero-Shot Text to Speech Synthesizers. arXiv preprint. https:\/\/www.microsoft.com\/en-us\/research\/publication\/vall-e-2-neural-codec-language-models-are-human-parity-zero-shot-text-to-speech-synthesizers-2\/"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888389"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1294"},{"key":"e_1_3_2_1_10_1","volume-title":"Retrieved","author":"Cohen Itai","year":"2025","unstructured":"Itai Cohen. 2024. The Evolution of Disinformation Campaigns: AI's Role in Creating Deepfakes. https:\/\/iamitcohen.medium.com\/the-evolution-of-disinformation-campaigns-ais-role-in-creating-deepfakes-074dac2cc431, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_11_1","volume-title":"TTS: A Deep Learning Toolkit for Text-to-Speech. https:\/\/github.com\/coqui-ai\/TTS. Accessed: 2025-04-08.","year":"2024","unstructured":"Coqui.ai. 2024. TTS: A Deep Learning Toolkit for Text-to-Speech. https:\/\/github.com\/coqui-ai\/TTS. Accessed: 2025-04-08."},{"key":"e_1_3_2_1_12_1","first-page":"5181","volume-title":"Proceedings of the 32nd USENIX Security Symposium (USENIX Security","author":"Deng Jiangyi","year":"2023","unstructured":"Jiangyi Deng, Fei Teng, Yanjiao Chen, Xiaofu Chen, Zhaohui Wang, and Wenyuan Xu. 2023. V-Cloak: Intelligibility-, Naturalness- & Timbre-Preserving Real-Time Voice Anonymization. In Proceedings of the 32nd USENIX Security Symposium (USENIX Security 2023). USENIX Association, 5181-5198. https:\/\/www.usenix.org\/conference\/usenixsecurity23\/presentation\/deng-jiangyi-v-cloak"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2019-28"},{"key":"e_1_3_2_1_15_1","volume-title":"Retrieved","author":"Finley Ben","year":"2025","unstructured":"Ben Finley. 2024. Deepfake of principal's voice is the latest case of AI being used for harm. https:\/\/apnews.com\/article\/ai-maryland-principal-voice-recording-663d5bc0714a3af221392cc6f1af985e, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_16_1","volume-title":"SpeechBrain: A General-Purpose Speech Toolkit. CoRR","author":"Gorodetskii Artem","year":"2022","unstructured":"Artem Gorodetskii and Ivan Ozhiganov. 2022. SpeechBrain: A General-Purpose Speech Toolkit. CoRR (2022). https:\/\/arxiv.org\/abs\/2201.10375"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"e_1_3_2_1_19_1","volume-title":"Advances in Neural Information Processing Systems 33 (NeurIPS","author":"Kim Jaehyeon","year":"2020","unstructured":"Jaehyeon Kim, Sungwon Kim, Jungil Kong, and Sungroh Yoon. 2020. Glow-TTS: A Generative Flow for Text-to-Speech via Monotonic Alignment Search. In Advances in Neural Information Processing Systems 33 (NeurIPS 2020). Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/5c3b99e8f92532e5ad1556e53ceea00c-Abstract.html"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCWC.2018.8301638"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413889"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/535"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016706"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2003"},{"key":"e_1_3_2_1_25_1","volume-title":"Retrieved","author":"AB.","year":"2025","unstructured":"Mizter_AB. 2023. Deepfake Technology and the Erosion of Personal Privacy. https:\/\/medium.com\/@abrahamedet9\/deepfake-technology-and-the-erosion-of-personal-privacy-4beb99e015f0, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(90)90021-Z"},{"key":"e_1_3_2_1_27_1","volume-title":"Retrieved","author":"Murf AI.","year":"2025","unstructured":"Murf AI. 2024. AI Voices: A Critical Gateway to Media and Entertainment Going Forward. https:\/\/murf.ai\/blog\/ai-voice-entertainment-media-tv, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-950"},{"key":"e_1_3_2_1_29_1","volume-title":"Retrieved","author":"NDTV.","year":"2025","unstructured":"NDTV. 2024. AI Scams Surge: Voice Cloning And Deepfake Threats Sweep India. https:\/\/www.ndtv.com\/ai\/ai-scams-surge-voice-cloning-and-deepfake-threats-sweep-india-6759260, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1977.1170350"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447871"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2307.08403"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-448"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.5555\/3618408.3619590"},{"key":"e_1_3_2_1_36_1","unstructured":"Mirco Ravanelli Titouan Parcollet Adel Moumen Sylvain de Langen Cem Subakan Peter Plantinga et al. 2024. Open-Source Conversational AI with SpeechBrain 1.0. CoRR (2024). https:\/\/arxiv.org\/abs\/2407.00463"},{"key":"e_1_3_2_1_37_1","unstructured":"Mirco Ravanelli Titouan Parcollet Peter Plantinga Aku Rouhe Samuele Cornell Loren Lugosch et al. 2021. SpeechBrain: A General-Purpose Speech Toolkit. CoRR (2021). https:\/\/arxiv.org\/abs\/2106.04624"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 9th International Conference on Learning Representations (ICLR","author":"Ren Yi","year":"2021","unstructured":"Yi Ren, Chenxu Hu, Xu Tan, Tao Qin, Sheng Zhao, Zhou Zhao, and Tie-Yan Liu. 2021. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. In Proceedings of the 9th International Conference on Learning Representations (ICLR 2021). OpenReview.net. https:\/\/openreview.net\/forum?id=piLPYqxtWuA"},{"key":"e_1_3_2_1_39_1","first-page":"3165","article-title":"FastSpeech: Fast, Robust and Controllable Text to Speech. In Advances in Neural Information Processing Systems 32 (NeurIPS 2019). Curran Associates","author":"Ren Yi","year":"2019","unstructured":"Yi Ren, Yangjun Ruan, Xu Tan, Tao Qin, Sheng Zhao, Zhou Zhao, and Tie-Yan Liu. 2019. FastSpeech: Fast, Robust and Controllable Text to Speech. In Advances in Neural Information Processing Systems 32 (NeurIPS 2019). Curran Associates, Inc., 3165-3174. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/f63f65b503e22cb970527f23c9ad7db1-Abstract.html","journal-title":"Inc."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.21437\/ICSLP.1992-125"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/MASS.2018.00016"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.21437\/INTERSPEECH.2015-92"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3038524"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCV.1995.477012"},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 19th International Society for Music Information Retrieval Conference (ISMIR 2018","author":"Stoller Daniel","year":"2018","unstructured":"Daniel Stoller, Sebastian Ewert, and Simon Dixon. 2018. Wave-U-Net: A Multi-Scale Neural Network for End-to-End Audio Source Separation. In Proceedings of the 19th International Society for Music Information Retrieval Conference (ISMIR 2018). 334-340. http:\/\/ismir2018.ircam.fr\/doc\/pdfs\/205_Paper.pdf"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495701"},{"key":"e_1_3_2_1_49_1","volume-title":"A Survey on Neural Speech Synthesis. CoRR","author":"Tan Xu","year":"2021","unstructured":"Xu Tan, Tao Qin, Frank K. Soong, and Tie-Yan Liu. 2021. A Survey on Neural Speech Synthesis. CoRR (2021). arXiv:2106.15561"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2867"},{"key":"e_1_3_2_1_51_1","volume-title":"Retrieved","author":"Vakulov Alex","year":"2025","unstructured":"Alex Vakulov. 2025. Deepfake Scams Are Stealing Millions-How To Spot One. https:\/\/www.forbes.com\/sites\/alexvakulov\/2025\/03\/09\/deepfake-scams-are-stealing-millions-how-to-spot-one\/, Retrieved April 8, 2025 from"},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the 9th ISCA Speech Synthesis Workshop (SSW 2016","author":"van den Oord A\u00e4ron","year":"2016","unstructured":"A\u00e4ron van den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew W. Senior, and Koray Kavukcuoglu. 2016. WaveNet: A Generative Model for Raw Audio. In Proceedings of the 9th ISCA Speech Synthesis Workshop (SSW 2016). ISCA, 125. https:\/\/www.isca-archive.org\/ssw_2016\/vandenoord16_ssw.html"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1513"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3558482.3590189"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CICTN57981.2023.10141447"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3407600"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3576915.3623209"},{"key":"e_1_3_2_1_59_1","volume-title":"BUT System Description to VoxCeleb Speaker Recognition Challenge","author":"Zeinali Hossein","year":"2019","unstructured":"Hossein Zeinali, Shuai Wang, Anna Silnova, Pavel Matejka, and Oldrich Plchot. 2019. BUT System Description to VoxCeleb Speaker Recognition Challenge 2019. CoRR (2019). http:\/\/arxiv.org\/abs\/1910.12592"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2303.13336"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3133956.3133962"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689217.3690615"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755629","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:57:36Z","timestamp":1765342656000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755629"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":62,"alternative-id":["10.1145\/3746027.3755629","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755629","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}