{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T20:10:18Z","timestamp":1765311018876,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755595","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:30:51Z","timestamp":1761377451000},"page":"11619-11628","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SiFMimicEvader: Evading Fake Voice Detection with Adversarial Neural Mimicry Attacks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0695-2187","authenticated-orcid":false,"given":"Xuan","family":"Hai","sequence":"first","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3685-4852","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5737-1301","authenticated-orcid":false,"given":"Zihao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8312-1881","authenticated-orcid":false,"given":"Ziyao","family":"Yu","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4829-3126","authenticated-orcid":false,"given":"Xiangzhen","family":"Kong","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7961-8502","authenticated-orcid":false,"given":"Song","family":"Li","sequence":"additional","affiliation":[{"name":"The State Key Laboratory of Blockchain and Data Security, Zhejiang University, Hangzhou, China and Hangzhou High-Tech Zone (Bin jiang) Institute of Blockchain and Data Security, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3235-3463","authenticated-orcid":false,"given":"Weina","family":"Niu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9968-6190","authenticated-orcid":false,"given":"Rui","family":"Zhou","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8054-5446","authenticated-orcid":false,"given":"Qingguo","family":"Zhou","sequence":"additional","affiliation":[{"name":"Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"CVPR workshops. 104-109","author":"AlBadawy Ehab A","year":"2019","unstructured":"Ehab A AlBadawy, Siwei Lyu, and Hany Farid. 2019. Detecting AI-Synthesized Speech Using Bispectral Analysis.. In CVPR workshops. 104-109."},{"key":"e_1_3_2_2_2_1","volume-title":"Deep residual neural networks for audio spoofing detection. arXiv preprint arXiv:1907.00501","author":"Alzantot Moustafa","year":"2019","unstructured":"Moustafa Alzantot, Ziqi Wang, and Mani B Srivastava. 2019. Deep residual neural networks for audio spoofing detection. arXiv preprint arXiv:1907.00501 (2019)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.115465"},{"key":"e_1_3_2_2_4_1","volume-title":"Science","volume":"364","author":"Bashivan Pouya","year":"2019","unstructured":"Pouya Bashivan, Kohitij Kar, and James J DiCarlo. 2019. Neural population control via deep image synthesis. Science, Vol. 364, 6439 (2019), eaav9436."},{"key":"e_1_3_2_2_5_1","first-page":"2691","volume-title":"31st USENIX Security Symposium (USENIX Security 22)","author":"Blue Logan","year":"2022","unstructured":"Logan Blue, Kevin Warren, Hadi Abdullah, Cassidy Gibson, Luis Vargas, Jessica O'Dell, Kevin Butler, and Patrick Traynor. 2022. Who are you (i really wanna know)? detecting audio {DeepFakes} through vocal tract reconstruction. In 31st USENIX Security Symposium (USENIX Security 22). 2691-2708."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3036777"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.26599\/BDMA.2022.9020017"},{"key":"e_1_3_2_2_8_1","volume-title":"Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979","author":"Conneau Alexis","year":"2020","unstructured":"Alexis Conneau, Alexei Baevski, Ronan Collobert, Abdelrahman Mohamed, and Michael Auli. 2020. Unsupervised cross-lingual representation learning for speech recognition. arXiv preprint arXiv:2006.13979 (2020)."},{"key":"e_1_3_2_2_9_1","unstructured":"Ingrid Daubechies et al. 2023. PyWavelets: Wavelet Transforms in Python. https:\/\/github.com\/PyWavelets\/pywt. Accessed on 2023-12-24."},{"key":"e_1_3_2_2_10_1","volume-title":"Xuechen Liu, Andreas Nautsch, Jose Patino, Md Sahidullah, Massimiliano Todisco, Xin Wang, and Others.","author":"Delgado H\u00e9ctor","year":"2021","unstructured":"H\u00e9ctor Delgado, Nicholas Evans, Tomi Kinnunen, Kong Aik Lee, Xuechen Liu, Andreas Nautsch, Jose Patino, Md Sahidullah, Massimiliano Todisco, Xin Wang, and Others. 2021. ASVspoof 2021: Automatic speaker verification spoofing and countermeasures challenge evaluation plan. arXiv preprint arXiv:2109.00535 (2021)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094704"},{"key":"e_1_3_2_2_12_1","volume-title":"Generalized spoofing detection inspired from audio generation artifacts. arXiv preprint arXiv:2104.04111","author":"Gao Yang","year":"2021","unstructured":"Yang Gao, Tyler Vuong, Mahsa Elyasi, Gaurav Bharaj, and Rita Singh. 2021. Generalized spoofing detection inspired from audio generation artifacts. arXiv preprint arXiv:2104.04111 (2021)."},{"key":"e_1_3_2_2_13_1","volume-title":"Raw differentiable architecture search for speech deepfake and spoofing detection. arXiv preprint arXiv:2107.12212","author":"Ge Wanying","year":"2021","unstructured":"Wanying Ge, Jose Patino, Massimiliano Todisco, and Nicholas Evans. 2021. Raw differentiable architecture search for speech deepfake and spoofing detection. arXiv preprint arXiv:2107.12212 (2021)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2212"},{"key":"e_1_3_2_2_15_1","first-page":"12702","article-title":"Audio Deepfake Detection With Self-Supervised Wavlm And Multi-Fusion Attentive Classifier. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Guo Yinlin","year":"2024","unstructured":"Yinlin Guo, Haofan Huang, Xi Chen, He Zhao, and Yuehai Wang. 2024. Audio Deepfake Detection With Self-Supervised Wavlm And Multi-Fusion Attentive Classifier. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 12702-12706.","journal-title":"IEEE"},{"key":"e_1_3_2_2_16_1","volume-title":"Ghost-in-Wave: How Speaker-Irrelative Features Interfere DeepFake Voice Detectors. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1-6.","author":"Hai Xuan","year":"2024","unstructured":"Xuan Hai, Xin Liu, Zhaorun Chen, Yuan Tan, Song Li, Weina Niu, Gang Liu, Rui Zhou, and Qingguo Zhou. 2024. Ghost-in-Wave: How Speaker-Irrelative Features Interfere DeepFake Voice Detectors. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1-6."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613841"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3089437"},{"key":"e_1_3_2_2_19_1","unstructured":"The Wall Street Journal. 2019. Fraudsters Used AI to Mimic CEO's Voice in Unusual Cybercrime Case. https:\/\/www.wsj.com\/articles\/fraudsters-use-ai-to-mimic ceos-voice-in-unusual-cybercrime-case-11567157402"},{"key":"e_1_3_2_2_20_1","volume-title":"ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, 6367-6371","author":"Heo Hee-Soo","year":"2022","unstructured":"Jee-weon Jung, Hee-Soo Heo, Hemlata Tak, Hye-jin Shim, Joon Son Chung, Bong-Jin Lee, Ha-Jin Yu, and Nicholas Evans. 2022. Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks. In ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, 6367-6371."},{"key":"e_1_3_2_2_21_1","volume-title":"Stargan-vc2: Rethinking conditional methods for stargan-based voice conversion. arXiv preprint arXiv:1907.12279","author":"Kaneko Takuhiro","year":"2019","unstructured":"Takuhiro Kaneko, Hirokazu Kameoka, Kou Tanaka, and Nobukatsu Hojo. 2019. Stargan-vc2: Rethinking conditional methods for stargan-based voice conversion. arXiv preprint arXiv:1907.12279 (2019)."},{"key":"e_1_3_2_2_22_1","volume-title":"Breaking Security-Critical Voice Authentication. In 2023 IEEE Symposium on Security and Privacy (SP). IEEE, 951-968","author":"Kassis Andre","year":"2023","unstructured":"Andre Kassis and Urs Hengartner. 2023. Breaking Security-Critical Voice Authentication. In 2023 IEEE Symposium on Security and Privacy (SP). IEEE, 951-968."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462693"},{"key":"e_1_3_2_2_24_1","volume-title":"STC antispoofing systems for the ASVspoof2019 challenge. arXiv preprint arXiv:1904.05576","author":"Lavrentyeva Galina","year":"2019","unstructured":"Galina Lavrentyeva, Sergey Novoselov, Andzhukaev Tseren, Marina Volkova, Artem Gorlanov, and Alexandr Kozlov. 2019. STC antispoofing systems for the ASVspoof2019 challenge. arXiv preprint arXiv:1904.05576 (2019)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413828"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053076"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003763"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSRE59848.2023.00029"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","unstructured":"Xin Liu Xiaokang Zhou and Qingguo Zhou. 2022. Human or Not: Can You Really Detect the Fake Voices? doi:10.13140\/RG.2.2.19572.83849","DOI":"10.13140\/RG.2.2.19572.83849"},{"key":"e_1_3_2_2_30_1","volume-title":"Speech is silver, silence is golden: What do ASVspoof-trained models really learn? arXiv preprint arXiv:2106.12914","author":"M\u00fcller Nicolas M","year":"2021","unstructured":"Nicolas M M\u00fcller, Franziska Dieckmann, Pavel Czempin, Roman Canals, Konstantin B\u00f6ttinger, and Jennifer Williams. 2021. Speech is silver, silence is golden: What do ASVspoof-trained models really learn? arXiv preprint arXiv:2106.12914 (2021)."},{"key":"e_1_3_2_2_31_1","unstructured":"NBC News. 2024. Fake Biden robocall telling Democrats not to vote is likely an AI-generated deepfake. https:\/\/www.nbcnews.com\/tech\/misinformation\/joe-biden-new-hampshire-robocall-fake-voice-deep-ai-primary-rcna135120"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_2_33_1","volume-title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers. arXiv preprint arXiv:2304.09116","author":"Shen Kai","year":"2023","unstructured":"Kai Shen, Zeqian Ju, Xu Tan, Yanqing Liu, Yichong Leng, Lei He, Tao Qin, Sheng Zhao, and Jiang Bian. 2023. Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers. arXiv preprint arXiv:2304.09116 (2023)."},{"key":"e_1_3_2_2_34_1","volume-title":"Graph attention networks for anti-spoofing. arXiv preprint arXiv:2104.03654","author":"Tak Hemlata","year":"2021","unstructured":"Hemlata Tak, Jee-weon Jung, Jose Patino, Massimiliano Todisco, and Nicholas Evans. 2021a. Graph attention networks for anti-spoofing. arXiv preprint arXiv:2104.03654 (2021)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414234"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2022-16"},{"key":"e_1_3_2_2_37_1","volume-title":"Naturalspeech: End-to-end text-to-speech synthesis with human-level quality","author":"Tan Xu","year":"2024","unstructured":"Xu Tan, Jiawei Chen, Haohe Liu, Jian Cong, Chen Zhang, Yanqing Liu, Xi Wang, Yichong Leng, Yuanhao Yi, Lei He, et al., 2024. Naturalspeech: End-to-end text-to-speech synthesis with human-level quality. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_2_38_1","volume-title":"Rohan Kumar Das, and Haizhou Li","author":"Tian Xiaohai","year":"2019","unstructured":"Xiaohai Tian, Rohan Kumar Das, and Haizhou Li. 2019. Black-box attacks on automatic speaker verification using feedback-controlled voice conversion. arXiv preprint arXiv:1909.07655 (2019)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746952"},{"key":"e_1_3_2_2_40_1","volume-title":"Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499","author":"Den Oord Aaron Van","year":"2016","unstructured":"Aaron Van Den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, Koray Kavukcuoglu, et al., 2016. Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499, Vol. 12 (2016)."},{"key":"e_1_3_2_2_41_1","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li et al. 2023. Neural codec language models are zero-shot text to speech synthesizers. arXiv preprint arXiv:2301.02111 (2023)."},{"key":"e_1_3_2_2_42_1","unstructured":"Hui Wang Siqi Zheng Yafeng Chen Luyao Cheng and Qian Chen. [n.d.]. CAM: A Fast and Efficient Network for Speaker Verification Using Context-Aware Masking. arXiv preprint arXiv:2303.00332 ( [n. d.])."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413716"},{"key":"e_1_3_2_2_44_1","volume-title":"Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135","author":"Wang Yuxuan","year":"2017","unstructured":"Yuxuan Wang, RJ Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron J Weiss, Navdeep Jaitly, Zongheng Yang, Ying Xiao, Zhifeng Chen, Samy Bengio, et al., 2017. Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135 (2017)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681345"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3306711"},{"key":"e_1_3_2_2_47_1","volume-title":"Mfa-conformer: Multi-scale feature aggregation conformer for automatic speaker verification. arXiv preprint arXiv:2203.15249","author":"Zhang Yang","year":"2022","unstructured":"Yang Zhang, Zhiqiang Lv, Haibin Wu, Shanshan Zhang, Pengfei Hu, Zhiyong Wu, Hung-yi Lee, and Helen Meng. 2022. Mfa-conformer: Multi-scale feature aggregation conformer for automatic speaker verification. arXiv preprint arXiv:2203.15249 (2022)."},{"key":"e_1_3_2_2_48_1","volume-title":"Proc. Interspeech.","author":"Yuxiang","year":"2021","unstructured":"Yuxiang Zhang12, Wenchao Wang12, and Pengyuan Zhang12. 2021. The effect of silence and dual-band fusion in anti-spoofing system. In Proc. Interspeech."},{"key":"e_1_3_2_2_49_1","first-page":"4840","article-title":"AdvTTS: Adversarial Text-to-Speech Synthesis Attack on Speaker Identification Systems. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Zuo Chu-Xiao","year":"2024","unstructured":"Chu-Xiao Zuo, Zhi-Jun Jia, and Wu-Jun Li. 2024. AdvTTS: Adversarial Text-to-Speech Synthesis Attack on Speaker Identification Systems. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 4840-4844.","journal-title":"IEEE"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755595","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T20:05:48Z","timestamp":1765310748000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755595"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":49,"alternative-id":["10.1145\/3746027.3755595","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755595","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}