{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:06:33Z","timestamp":1765343193983,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23B2022, U22B2047, and 62202310"],"award-info":[{"award-number":["U23B2022, U22B2047, and 62202310"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guangdong Basic and Applied Basic Research Foundation","award":["2025A1515010292"],"award-info":[{"award-number":["2025A1515010292"]}]},{"name":"Guangdong Provincial Key Laboratory","award":["2023B1212060076"],"award-info":[{"award-number":["2023B1212060076"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754741","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:55Z","timestamp":1761377215000},"page":"7277-7286","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ALDEN: Dual-Level Disentanglement with Meta-learning for Generalizable Audio Deepfake Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0514-3698","authenticated-orcid":false,"given":"Yuxiong","family":"Xu","sequence":"first","affiliation":[{"name":"Guangdong Provincial Key Laboratory of Intelligent Information Processing, Shenzhen, China, Shenzhen Key Laboratory of Media Security, Shenzhen, China, and Shenzhen University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2613-5451","authenticated-orcid":false,"given":"Bin","family":"Li","sequence":"additional","affiliation":[{"name":"Guangdong Provincial Key Laboratory of Intelligent Information Processing, Shenzhen, China, Shenzhen Key Laboratory of Media Security, Shenzhen, China, and Shenzhen University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5094-6548","authenticated-orcid":false,"given":"Weixiang","family":"Li","sequence":"additional","affiliation":[{"name":"Guangdong Provincial Key Laboratory of Intelligent Information Processing, Shenzhen, China, Shenzhen Key Laboratory of Media Security, Shenzhen, China, and Shenzhen University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3811-003X","authenticated-orcid":false,"given":"Sara","family":"Mandelli","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Dipartimento di Elettronica, Informazione e Bioingegneria (DEIB), Milan, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2483-4366","authenticated-orcid":false,"given":"Viola","family":"Negroni","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Dipartimento di Elettronica, Informazione e Bioingegneria (DEIB), Milan, Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2761-1918","authenticated-orcid":false,"given":"Sheng","family":"Li","sequence":"additional","affiliation":[{"name":"Afirstsoft Technology Group Co., Ltd., Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems, Vol. 33 (2020), 12449-12460.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26479"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 340-348","author":"Bei Yijun","year":"2024","unstructured":"Yijun Bei, Xing Zhou, Erteng Liu, Yang Gao, Sen Lin, Kewei Gao, and Zunlei Feng. 2024. Discriminative feature decoupling enhancement for speech forgery detection. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 340-348."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.50"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2010.764"},{"key":"e_1_3_2_1_6_1","volume-title":"Adaptive Mixture of Low-Rank Experts for Robust Audio Spoofing Detection. arXiv:2503.12010","author":"Chen Qixian","year":"2025","unstructured":"Qixian Chen, Yuxiong Xu, Sara Mandelli, Sheng Li, and Bin Li. 2025. Adaptive Mixture of Low-Rank Experts for Robust Audio Spoofing Detection. arXiv:2503.12010 (2025)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.21437\/ASVSPOOF.2021-14"},{"key":"e_1_3_2_1_9_1","first-page":"664","article-title":"One-Shot Voice Conversion by Separating Speaker and Content Representations with Instance Normalization","volume":"2019","author":"Hung-Yi Lee Chou","year":"2019","unstructured":"Ju-chieh Chou and Hung-Yi Lee. 2019. One-Shot Voice Conversion by Separating Speaker and Content Representations with Instance Normalization. In Proc. Interspeech 2019. 664-668.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143874"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1.","author":"Frank Joel","year":"2021","unstructured":"Joel Frank and Lea Sch\u00f6nherr. 2021. Wavefake: A data set to facilitate audio deepfake detection. In Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1."},{"key":"e_1_3_2_1_12_1","first-page":"376","article-title":"Audio Anti-spoofing Using Simple Attention Module and Joint Optimization Based on Additive Angular Margin Loss and Meta-learning","volume":"2022","author":"Hansen John HL","year":"2022","unstructured":"John HL Hansen and Zhenyu Wang. 2022. Audio Anti-spoofing Using Simple Attention Module and Joint Optimization Based on Additive Angular Margin Loss and Meta-learning. In Proc. Interspeech 2022. 376-380.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.167"},{"key":"e_1_3_2_1_14_1","volume-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6367-6371","author":"Heo Hee-Soo","year":"2022","unstructured":"Jee-weon Jung, Hee-Soo Heo, Hemlata Tak, Hye-jin Shim, Joon Son Chung, Bong-Jin Lee, Ha-Jin Yu, and Nicholas Evans. 2022. Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6367-6371."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-024-10922-z"},{"key":"e_1_3_2_1_16_1","first-page":"17022","article-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"Kong Jungil","year":"2020","unstructured":"Jungil Kong, Jaehyeon Kim, and Jaekyoung Bae. 2020. Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis. Advances in Neural Information Processing Systems, Vol. 33 (2020), 17022-17033.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_17_1","volume-title":"Meta-Learning Approaches For Improving Detection of Unseen Speech Deepfakes. In 2024 IEEE Spoken Language Technology Workshop (SLT). 1173-1178","author":"Kukanov Ivan","year":"2024","unstructured":"Ivan Kukanov, Janne Laakkonen, Tomi Kinnunen, and Ville Hautam\u00e4ki. 2024. Meta-Learning Approaches For Improving Detection of Unseen Speech Deepfakes. In 2024 IEEE Spoken Language Technology Workshop (SLT). 1173-1178."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1768"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11596"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630751"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095191"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3670285"},{"key":"e_1_3_2_1_23_1","first-page":"1","article-title":"EAD-VC: Enhancing Speech Auto-Disentanglement for Voice Conversion with IFUB Estimator and Joint Text-Guided Consistent Learning. In 2024 International Joint Conference on Neural Networks (IJCNN)","author":"Liang Ziqi","year":"2024","unstructured":"Ziqi Liang, Jianzong Wang, Xulong Zhang, Yong Zhang, Ning Cheng, and Jing Xiao. 2024. EAD-VC: Enhancing Speech Auto-Disentanglement for Voice Conversion with IFUB Estimator and Joint Text-Guided Consistent Learning. In 2024 International Joint Conference on Neural Networks (IJCNN). IEEE, 1-7.","journal-title":"IEEE"},{"key":"e_1_3_2_1_24_1","volume-title":"ASSD: An AI-Synthesized Speech Detection Scheme Using Whisper Feature and Types Classification","author":"Liu Chang","year":"2025","unstructured":"Chang Liu, Xiaolong Xu, and Fu Xiao. 2025. ASSD: An AI-Synthesized Speech Detection Scheme Using Whisper Feature and Types Classification. IEEE Transactions on Audio, Speech and Language Processing (2025)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3285283"},{"key":"e_1_3_2_1_26_1","volume-title":"Challenge. In ICASSP, 2022-2022 IEEE International Conference, on Acoustics, Speech, and Signal Processing (ICASSP, ). IEEE, 9241-9245","author":"Juan M.","year":"2022","unstructured":"Juan M. Mart\u00edn-Do nas, and Aitor \u00c1lvarez. 2022. The Vicomtech Audio Deepfake Detection System Based on Wav2vec2 for the 2022 ADD, Challenge. In ICASSP, 2022-2022 IEEE International Conference, on Acoustics, Speech, and Signal Processing (ICASSP, ). IEEE, 9241-9245."},{"key":"e_1_3_2_1_27_1","first-page":"2783","article-title":"Does Audio Deepfake Detection Generalize?","volume":"2022","author":"M\u00fcller Nicolas","year":"2022","unstructured":"Nicolas M\u00fcller, Pavel Czempin, Franziska Diekmann, Adam Froghyar, and Konstantin B\u00f6ttinger. 2022. Does Audio Deepfake Detection Generalize?. In Proc. Interspeech 2022. 2783-2787.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_28_1","volume-title":"Synthetic speech detection using meta-learning with prototypical loss. arXiv:2201.09470","author":"Pal Monisankha","year":"2022","unstructured":"Monisankha Pal, Aditya Raikar, Ashish Panda, and Sunil Kumar Kopparapu. 2022. Synthetic speech detection using meta-learning with prototypical loss. arXiv:2201.09470 (2022)."},{"key":"e_1_3_2_1_29_1","volume-title":"A comprehensive survey with critical analysis for deepfake speech detection. arXiv:2409.15180","author":"Pham Lam","year":"2024","unstructured":"Lam Pham, Phat Lam, Dat Tran, Hieu Tang, Tin Nguyen, Alexander Schindler, Florian Skopik, Alexander Polonsky, and Canh Vu. 2024. A comprehensive survey with critical analysis for deepfake speech detection. arXiv:2409.15180 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"e_1_3_2_1_31_1","volume-title":"Improving Generalization for AI-Synthesized Voice Detection. arXiv:2412.19279","author":"Ren Hainan","year":"2024","unstructured":"Hainan Ren, Li Lin, Chun-Hao Liu, Xin Wang, and Shu Hu. 2024. Improving Generalization for AI-Synthesized Voice Detection. arXiv:2412.19279 (2024)."},{"key":"e_1_3_2_1_32_1","first-page":"5281","article-title":"A conformer-based classifier for variable-length utterance processing in anti-spoofing","volume":"2023","author":"Rosello Eros","year":"2023","unstructured":"Eros Rosello, Alejandro Gomez-Alanis, Angel M Gomez, and Antonio Peinado. 2023. A conformer-based classifier for variable-length utterance processing in anti-spoofing. In Proc. Interspeech 2023. 5281-5285.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_33_1","first-page":"2087","article-title":"A comparison of features for synthetic speech detection","volume":"2015","author":"Sahidullah Md.","year":"2015","unstructured":"Md. Sahidullah, Tomi Kinnunen, and Cemal Hanil\u00e7i. 2015. A comparison of features for synthetic speech detection. In Proc. Interspeech 2015. 2087-2091.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00097"},{"key":"e_1_3_2_1_35_1","first-page":"2356","article-title":"Graph Attention Networks for Anti-Spoofing","volume":"2021","author":"Tak Hemlata","year":"2021","unstructured":"Hemlata Tak, Jee-weon Jung, Jose Patino, Massimiliano Todisco, and Nicholas W. D. Evans. 2021. Graph Attention Networks for Anti-Spoofing. In Proc. Interspeech 2021. 2356-2360.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_36_1","unstructured":"Hemlata Tak Massimiliano Todisco Xin Wang Jee-weon Jung Junichi Yamagishi and Nicholas WD Evans. 2022. Automatic Speaker Verification Spoofing and Deepfake Detection Using Wav2vec 2.0 and Data Augmentation. In Odyssey."},{"key":"e_1_3_2_1_37_1","volume-title":"A survey on neural speech synthesis. arXiv:2106.15561","author":"Tan Xu","year":"2021","unstructured":"Xu Tan, Tao Qin, Frank Soong, and Tie-Yan Liu. 2021. A survey on neural speech synthesis. arXiv:2106.15561 (2021)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2017.01.001"},{"key":"e_1_3_2_1_39_1","volume-title":"ASVspoof 2019: Future horizons in spoofed and fake audio detection. arXiv:1904.05441","author":"Todisco Massimiliano","year":"2019","unstructured":"Massimiliano Todisco, Xin Wang, Ville Vestman, Md Sahidullah, H\u00e9ctor Delgado, Andreas Nautsch, Junichi Yamagishi, Nicholas Evans, Tomi Kinnunen, and Kong Aik Lee. 2019. ASVspoof 2019: Future horizons in spoofed and fake audio detection. arXiv:1904.05441 (2019)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2022.101362"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3414340"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3420937"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2956145"},{"key":"e_1_3_2_1_44_1","first-page":"1","article-title":"Spoofed Training Data for Speech Spoofing Countermeasure Can Be Efficiently Created Using Neural Vocoders. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Wang Xin","year":"2023","unstructured":"Xin Wang and Junichi Yamagishi. 2023. Spoofed Training Data for Speech Spoofing Countermeasure Can Be Efficiently Created Using Neural Vocoders. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5.","journal-title":"IEEE"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3421281"},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the 39th International Conference, on Machine Learning. PMLR, 23631-23644","author":"Wei Hongxin","year":"2022","unstructured":"Hongxin Wei, Renchunzi Xie, Hao Cheng, Lei Feng, Bo An, and Yixuan Li. 2022. Mitigating Neural Network Overconfidence, with Logit Normalization. In Proceedings of the 39th International Conference, on Machine Learning. PMLR, 23631-23644."},{"key":"e_1_3_2_1_47_1","unstructured":"Yuankun Xie Yi Lu Ruibo Fu Zhengqi Wen Zhiyong Wang Jianhua Tao Xin Qi Xiaopeng Wang Yukun Liu Haonan Cheng et al. 2025. The codecfake dataset and countermeasures for the universally detection of deepfake audio. IEEE Transactions on Audio Speech and Language Processing (2025)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.11834\/jig.230476"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.21437\/ASVspoof.2024-10"},{"key":"e_1_3_2_1_50_1","volume-title":"Dsvae: Interpretable disentangled representation for synthetic speech detection. arXiv:2304.03323","author":"Singh Yadav Amit Kumar","year":"2023","unstructured":"Amit Kumar Singh Yadav, Kratika Bhagtani, Ziyue Xiang, Paolo Bestagini, Stefano Tubaro, and Edward J Delp. 2023. Dsvae: Interpretable disentangled representation for synthetic speech detection. arXiv:2304.03323 (2023)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552466.3556525"},{"key":"e_1_3_2_1_52_1","first-page":"1998","article-title":"Improving Generalization Ability of Countermeasures for New Mismatch Scenario by Combining Multiple Advanced Regularization Terms","volume":"2023","author":"Zeng Chang","year":"2023","unstructured":"Chang Zeng, Xin Wang, Xiaoxiao Miao, Erica Cooper, and Junichi Yamagishi. 2023. Improving Generalization Ability of Countermeasures for New Mismatch Scenario by Combining Multiple Advanced Regularization Terms. In Proc. Interspeech 2023. 1998-2002.","journal-title":"Proc. Interspeech"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2024.3520001"},{"key":"e_1_3_2_1_54_1","first-page":"6","article-title":"Recall, precision and average precision. Department of Statistics and Actuarial Science, University of Waterloo","volume":"2","author":"Zhu Mu","year":"2004","unstructured":"Mu Zhu. 2004. Recall, precision and average precision. Department of Statistics and Actuarial Science, University of Waterloo, Waterloo, Vol. 2, 30 (2004), 6.","journal-title":"Waterloo"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754741","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:03:23Z","timestamp":1765343003000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754741"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":54,"alternative-id":["10.1145\/3746027.3754741","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754741","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}