{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:03:21Z","timestamp":1750309401052,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681100","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"6900-6909","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["What's the Real: A Novel Design Philosophy for Robust AI-Synthesized Voice Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0695-2187","authenticated-orcid":false,"given":"Xuan","family":"Hai","sequence":"first","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3685-4852","authenticated-orcid":false,"given":"Xin","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3260-6771","authenticated-orcid":false,"given":"Yuan","family":"Tan","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7139-0462","authenticated-orcid":false,"given":"Gang","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7961-8502","authenticated-orcid":false,"given":"Song","family":"Li","sequence":"additional","affiliation":[{"name":"The State Key Laboratory of Blockchain and Data Security, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3235-3463","authenticated-orcid":false,"given":"Weina","family":"Niu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9968-6190","authenticated-orcid":false,"given":"Rui","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3488-4679","authenticated-orcid":false,"given":"Xiaokang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Faculty of Business Data Science, Kansai University, Osaka, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2021. ASVspoof 2021: Automatic Speaker Verification Spoofing and Countermeasures Challenge. https:\/\/www.asvspoof.org\/index2021.html Accessed: 2024-3-31."},{"key":"e_1_3_2_1_2_1","volume-title":"CVPR workshops. 104--109","author":"AlBadawy Ehab A","year":"2019","unstructured":"Ehab A AlBadawy, Siwei Lyu, and Hany Farid. 2019. Detecting AI-Synthesized Speech Using Bispectral Analysis.. In CVPR workshops. 104--109."},{"key":"e_1_3_2_1_3_1","volume-title":"Deep residual neural networks for audio spoofing detection. arXiv preprint arXiv:1907.00501","author":"Alzantot Moustafa","year":"2019","unstructured":"Moustafa Alzantot, ZiqiWang, and Mani B Srivastava. 2019. Deep residual neural networks for audio spoofing detection. arXiv preprint arXiv:1907.00501 (2019)."},{"key":"e_1_3_2_1_4_1","volume-title":"International conference on machine learning. PMLR, 195--204","author":"Ar\u0131k Sercan","year":"2017","unstructured":"Sercan \u00d6 Ar\u0131k, Mike Chrzanowski, Adam Coates, Gregory Diamos, Andrew Gibiansky, Yongguo Kang, Xian Li, John Miller, Andrew Ng, Jonathan Raiman, et al. 2017. Deep voice: Real-time neural text-to-speech. In International conference on machine learning. PMLR, 195--204."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Arun Babu Changhan Wang Andros Tjandra Kushal Lakhotia Qiantong Xu Naman Goyal Kritika Singh Patrick von Platen Yatharth Saraf Juan Pino et al. 2021. XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:2111.09296 (2021).","DOI":"10.21437\/Interspeech.2022-143"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.115465"},{"key":"e_1_3_2_1_7_1","volume-title":"31st USENIX Security Symposium (USENIX Security 22)","author":"Blue Logan","year":"2022","unstructured":"Logan Blue, Kevin Warren, Hadi Abdullah, Cassidy Gibson, Luis Vargas, Jessica O'Dell, Kevin Butler, and Patrick Traynor. 2022. Who are you (i really wanna know)? detecting audio {DeepFakes} through vocal tract reconstruction. In 31st USENIX Security Symposium (USENIX Security 22). 2691--2708."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3036777"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2018-42"},{"key":"e_1_3_2_1_10_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_11_1","volume-title":"SAMO: Speaker Attractor Multi-Center One-Class Learning For Voice Anti-Spoofing. In ICASSP 2023--2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1--5.","author":"Ding Siwen","year":"2023","unstructured":"Siwen Ding, You Zhang, and Zhiyao Duan. 2023. SAMO: Speaker Attractor Multi-Center One-Class Learning For Voice Anti-Spoofing. In ICASSP 2023--2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1--5."},{"volume-title":"FastAudio: A Learnable Audio Front-End for Spoof Speech Detection. In 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE.","author":"Fu Quchen","key":"e_1_3_2_1_12_1","unstructured":"Quchen Fu, Zhongwei Teng, Jules White, M. Powell, and Douglas C. Schmidt. 2022. FastAudio: A Learnable Audio Front-End for Spoof Speech Detection. In 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE."},{"key":"e_1_3_2_1_13_1","first-page":"1068","article-title":"A light convolutional GRU-RNN deep feature extractor for ASV spoofing detection","volume":"2019","author":"Gomez-Alanis Alejandro","year":"2019","unstructured":"Alejandro Gomez-Alanis, Antonio M Peinado, Jose A Gonzalez, and Angel M Gomez. 2019. A light convolutional GRU-RNN deep feature extractor for ASV spoofing detection. In Proc. Interspeech, Vol. 2019. 1068--1072.","journal-title":"Proc. Interspeech"},{"volume-title":"ICASSP 2024--2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Guo Yinlin","key":"e_1_3_2_1_14_1","unstructured":"Yinlin Guo, Haofan Huang, Xi Chen, He Zhao, and Yuehai Wang. 2024. Audio Deepfake Detection With Self-Supervised Wavlm And Multi-Fusion Attentive Classifier. In ICASSP 2024--2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 12702--12706."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613841"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2016.7820786"},{"key":"e_1_3_2_1_17_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu Edward J","year":"2021","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)."},{"key":"e_1_3_2_1_18_1","unstructured":"The Wall Street Journal. 2019. Fraudsters Used AI to Mimic CEO?s Voice in Unusual Cybercrime Case. https:\/\/www.wsj.com\/articles\/fraudsters-use-ai-tomimicceos-voice-in-unusual-cybercrime-case-11567157402 ."},{"key":"e_1_3_2_1_19_1","volume-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6367--6371","author":"Heo Hee-Soo","year":"2022","unstructured":"Jee-weon Jung, Hee-Soo Heo, Hemlata Tak, Hye-jin Shim, Joon Son Chung, Bong-Jin Lee, Ha-Jin Yu, and Nicholas Evans. 2022. Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6367--6371."},{"key":"e_1_3_2_1_20_1","volume-title":"Parallel-data-free voice conversion using cycle-consistent adversarial networks. arXiv preprint arXiv:1711.11293","author":"Kaneko Takuhiro","year":"2017","unstructured":"Takuhiro Kaneko and Hirokazu Kameoka. 2017. Parallel-data-free voice conversion using cycle-consistent adversarial networks. arXiv preprint arXiv:1711.11293 (2017)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682897"},{"key":"e_1_3_2_1_22_1","volume-title":"Breaking Security-Critical Voice Authentication. In 2023 IEEE Symposium on Security and Privacy (SP). IEEE, 951--968","author":"Kassis Andre","year":"2023","unstructured":"Andre Kassis and Urs Hengartner. 2023. Breaking Security-Critical Voice Authentication. In 2023 IEEE Symposium on Security and Privacy (SP). IEEE, 951--968."},{"key":"e_1_3_2_1_23_1","volume-title":"Improved DeepFake Detection Using Whisper Features. arXiv preprint arXiv:2306.01428","author":"Kawa Piotr","year":"2023","unstructured":"Piotr Kawa, Marcin Plata, Micha\u0140 Czuba, Piotr Szyma\u0144ski, and Piotr Syga. 2023. Improved DeepFake Detection Using Whisper Features. arXiv preprint arXiv:2306.01428 (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"International Conference on Machine Learning. PMLR, 5530--5540","author":"Kim Jaehyeon","year":"2021","unstructured":"Jaehyeon Kim, Jungil Kong, and Juhee Son. 2021. Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In International Conference on Machine Learning. PMLR, 5530--5540."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSRE59848.2023.00029"},{"key":"e_1_3_2_1_26_1","unstructured":"Xuechen Liu Xin Wang Md Sahidullah Jose Patino H\u00e9ctor Delgado Tomi Kinnunen Massimiliano Todisco Junichi Yamagishi Nicholas Evans Andreas Nautsch et al. 2023. Asvspoof 2021: Towards spoofed and deepfake speech detection in the wild. IEEE\/ACM Transactions on Audio Speech and Language Processing (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","unstructured":"Xin Liu Xiaokang Zhou and Qingguo Zhou. 2022. Human or Not: Can You Really Detect the Fake Voices? https:\/\/doi.org\/10.13140\/RG.2.2.19572.83849","DOI":"10.13140\/RG.2.2.19572.83849"},{"key":"e_1_3_2_1_28_1","volume-title":"Multi-Task Learning Improves Synthetic Speech Detection. In ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 6392--6396","author":"Mo Yichuan","year":"2022","unstructured":"Yichuan Mo and Shilin Wang. 2022. Multi-Task Learning Improves Synthetic Speech Detection. In ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 6392--6396. https:\/\/doi.org\/10.1109\/ ICASSP43922.2022.9746059"},{"key":"e_1_3_2_1_29_1","volume-title":"Speech is silver, silence is golden: What do ASVspoof-trained models really learn? arXiv preprint arXiv:2106.12914","author":"M\u00fcller Nicolas M","year":"2021","unstructured":"Nicolas M M\u00fcller, Franziska Dieckmann, Pavel Czempin, Roman Canals, Konstantin B\u00f6ttinger, and Jennifer Williams. 2021. Speech is silver, silence is golden: What do ASVspoof-trained models really learn? arXiv preprint arXiv:2106.12914 (2021)."},{"key":"e_1_3_2_1_30_1","unstructured":"NBC News. 2024. Fake Biden robocall telling Democrats not to vote is likely an AI-generated deepfake. https:\/\/www.nbcnews.com\/tech\/misinformation\/joebiden-new-hampshire-robocall-fake-voice-deep-ai-primary-rcna135120"},{"key":"e_1_3_2_1_31_1","volume-title":"Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499","author":"van den Oord Aaron","year":"2016","unstructured":"Aaron van den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, and Koray Kavukcuoglu. 2016. Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 (2016)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"e_1_3_2_1_33_1","volume-title":"Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers. arXiv preprint arXiv:2304.09116","author":"Shen Kai","year":"2023","unstructured":"Kai Shen, Zeqian Ju, Xu Tan, Yanqing Liu, Yichong Leng, Lei He, Tao Qin, Sheng Zhao, and Jiang Bian. 2023. Naturalspeech 2: Latent diffusion models are natural and zero-shot speech and singing synthesizers. arXiv preprint arXiv:2304.09116 (2023)."},{"key":"e_1_3_2_1_34_1","volume-title":"End-to-end spectro-temporal graph attention networks for speaker verification anti-spoofing and speech deepfake detection. arXiv preprint arXiv:2107.12710","author":"Tak Hemlata","year":"2021","unstructured":"Hemlata Tak, Jee-weon Jung, Jose Patino, Madhu Kamble, Massimiliano Todisco, and Nicholas Evans. 2021. End-to-end spectro-temporal graph attention networks for speaker verification anti-spoofing and speech deepfake detection. arXiv preprint arXiv:2107.12710 (2021)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414234"},{"key":"e_1_3_2_1_36_1","volume-title":"Naturalspeech: End-to-end text-tospeech synthesis with human-level quality","author":"Tan Xu","year":"2024","unstructured":"Xu Tan, Jiawei Chen, Haohe Liu, Jian Cong, Chen Zhang, Yanqing Liu, Xi Wang, Yichong Leng, Yuanhao Yi, Lei He, et al. 2024. Naturalspeech: End-to-end text-tospeech synthesis with human-level quality. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472051"},{"key":"e_1_3_2_1_38_1","volume-title":"Md Sahidullah, Nicholas Evans, Tomi Kinnunen, and Junichi Yamagishi.","author":"Todisco Massimiliano","year":"2018","unstructured":"Massimiliano Todisco, H\u00e9ctor Delgado, Kong Aik Lee, Md Sahidullah, Nicholas Evans, Tomi Kinnunen, and Junichi Yamagishi. 2018. Integrated presentation attack detection and automatic speaker verification: Common features and gaussian back-end fusion. In Interspeech 2018-19th Annual Conference of the International Speech Communication Association. ISCA."},{"key":"e_1_3_2_1_39_1","volume-title":"Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 12","author":"Den Oord Aaron Van","year":"2016","unstructured":"Aaron Van Den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, Koray Kavukcuoglu, et al. 2016. Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 12 (2016)."},{"key":"e_1_3_2_1_40_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_41_1","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li et al. 2023. Neural codec language models are zero-shot text to speech synthesizers. arXiv preprint arXiv:2301.02111 (2023)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413716"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101114"},{"key":"e_1_3_2_1_44_1","volume-title":"Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135","author":"Wang Yuxuan","year":"2017","unstructured":"Yuxuan Wang, RJ Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron J Weiss, Navdeep Jaitly, Zongheng Yang, Ying Xiao, Zhifeng Chen, Samy Bengio, et al. 2017. Tacotron: Towards end-to-end speech synthesis. arXiv preprint arXiv:1703.10135 (2017)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413851"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3306711"},{"key":"e_1_3_2_1_47_1","volume-title":"Proc. Interspeech.","author":"Yuxiang","year":"2021","unstructured":"Yuxiang Zhang12, Wenchao Wang12, and Pengyuan Zhang12. 2021. The effect of silence and dual-band fusion in anti-spoofing system. In Proc. Interspeech."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681100","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681100","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:52Z","timestamp":1750294672000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681100"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":47,"alternative-id":["10.1145\/3664647.3681100","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681100","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}