{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T15:31:44Z","timestamp":1784043104193,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472032"],"award-info":[{"award-number":["62472032"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Young Elite Scientists Sponsorship Program by CAST","award":["2023QNRC001"],"award-info":[{"award-number":["2023QNRC001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755783","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:17Z","timestamp":1761375257000},"page":"8910-8919","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["SepVAMark: Deep Separable Visual-Audio Fusion Watermarking for Source Tracing and Deepfake Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7684-8540","authenticated-orcid":false,"given":"Chuan","family":"Zhang","sequence":"first","affiliation":[{"name":"Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4765-8244","authenticated-orcid":false,"given":"Zihan","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology, Beijjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2612-3661","authenticated-orcid":false,"given":"Zihao","family":"Xu","sequence":"additional","affiliation":[{"name":"Changchun University, Beijjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7253-0276","authenticated-orcid":false,"given":"Xuhao","family":"Ren","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3277-3887","authenticated-orcid":false,"given":"Liehuang","family":"Zhu","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WIFS.2018.8630761"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.3390\/a15050155"},{"key":"e_1_3_2_1_3_1","volume-title":"Audio-visual Deepfake Detection With Local Temporal Inconsistencies. arXiv preprint arXiv:2501.08137","author":"Astrid Marcella","year":"2025","unstructured":"Marcella Astrid, Enjie Ghorbel, and Djamila Aouada. 2025. Audio-visual Deepfake Detection With Local Temporal Inconsistencies. arXiv preprint arXiv:2501.08137 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Frequency domain-based detection of generated audio. arXiv preprint arXiv:2205.01806","author":"Bartusiak Emily R","year":"2022","unstructured":"Emily R Bartusiak and Edward J Delp. 2022. Frequency domain-based detection of generated audio. arXiv preprint arXiv:2205.01806 (2022)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00435"},{"key":"e_1_3_2_1_6_1","volume-title":"Wavmark: Watermarking for audio generation. arXiv preprint arXiv:2308.12770","author":"Chen Guangyu","year":"2023","unstructured":"Guangyu Chen, Yu Wu, Shujie Liu, Tao Liu, Xiaoyong Du, and Furu Wei. 2023. Wavmark: Watermarking for audio generation. arXiv preprint arXiv:2308.12770 (2023)."},{"key":"e_1_3_2_1_7_1","first-page":"2003","volume-title":"SimSwap: An Efficient Framework For High Fidelity Face Swapping. In MM '20: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA","author":"Chen Renwang","year":"2020","unstructured":"Renwang Chen, Xuanhong Chen, Bingbing Ni, and Yanhao Ge. 2020. SimSwap: An Efficient Framework For High Fidelity Face Swapping. In MM '20: The 28th ACM International Conference on Multimedia, Virtual Event \/ Seattle, WA, USA, October 12-16, 2020, Chang Wen Chen, Rita Cucchiara, Xian-Sheng Hua, Guo-Jun Qi, Elisa Ricci, Zhengyou Zhang, and Roger Zimmermann (Eds.). ACM, 2003-2011."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00916"},{"key":"e_1_3_2_1_9_1","unstructured":"Flavelle Christopher Nehamas Nicholas Ngo Madeleine and Demirjian Karoun. [n.d.]. Fake Video of Trump and Musk Appears on TVs at Housing Agency. [Online]. https:\/\/www.nytimes.com\/2025\/02\/24\/us\/politics\/musk-trump-toes-video.html\/."},{"key":"e_1_3_2_1_10_1","first-page":"1","article-title":"AVSecure: An Audio-Visual Watermarking Framework for Proactive Deepfake Detection. In 2024 IEEE 14th International Conference on Electronics Information and Emergency Communication (ICEIEC)","author":"Guo Bofei","year":"2024","unstructured":"Bofei Guo, Haoxuan Tai, Guibo Luo, and Yuesheng Zhu. 2024. AVSecure: An Audio-Visual Watermarking Framework for Proactive Deepfake Detection. In 2024 IEEE 14th International Conference on Electronics Information and Emergency Communication (ICEIEC). IEEE, 1-4.","journal-title":"IEEE"},{"key":"e_1_3_2_1_11_1","volume-title":"Generalizable Audio Deepfake Detection via Latent Space Refinement and Augmentation. arXiv preprint arXiv:2501.14240","author":"Huang Wen","year":"2025","unstructured":"Wen Huang, Yanmei Gu, Zhiming Wang, Huijia Zhu, and Yanmin Qian. 2025. Generalizable Audio Deepfake Detection via Latent Space Refinement and Augmentation. arXiv preprint arXiv:2501.14240 (2025)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_2_1_13_1","volume-title":"Yonghui Wu, et al.","author":"Jia Ye","year":"2018","unstructured":"Ye Jia, Yu Zhang, Ron Weiss, Quan Wang, Jonathan Shen, Fei Ren, Patrick Nguyen, Ruoming Pang, Ignacio Lopez Moreno, Yonghui Wu, et al., 2018. Transfer learning from speaker verification to multispeaker text-to-speech synthesis. Advances in neural information processing systems, Vol. 31 (2018)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475324"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2988660"},{"key":"e_1_3_2_1_16_1","volume-title":"FakeAVCeleb: A novel audio-video multimodal deepfake dataset. arXiv preprint arXiv:2108.05080","author":"Khalid Hasam","year":"2021","unstructured":"Hasam Khalid, Shahroz Tariq, Minha Kim, and Simon S Woo. 2021. FakeAVCeleb: A novel audio-video multimodal deepfake dataset. arXiv preprint arXiv:2108.05080 (2021)."},{"key":"e_1_3_2_1_17_1","volume-title":"A deep learning framework for audio deepfake detection. Arabian Journal for Science and Engineering","author":"Khochare Janavi","year":"2021","unstructured":"Janavi Khochare, Chaitali Joshi, Bakul Yenarkar, Shraddha Suratkar, and Faruk Kazi. 2021. A deep learning framework for audio deepfake detection. Arabian Journal for Science and Engineering (2021), 1-12."},{"key":"e_1_3_2_1_18_1","volume-title":"Chenquan Gan, and Xudong Zhao.","author":"Kumar Ashish","year":"2025","unstructured":"Ashish Kumar, Divya Singh, Rachna Jain, Deepak Kumar Jain, Chenquan Gan, and Xudong Zhao. 2025. Advances in DeepFake detection algorithms: Exploring fusion techniques in single and multi-modal approach. Information Fusion (2025), 102993."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2020.107616"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9413139"},{"key":"e_1_3_2_1_21_1","volume-title":"IDEAW: Robust Neural Audio Watermarking with Invertible Dual-Embedding. arXiv preprint arXiv:2409.19627","author":"Li Pengcheng","year":"2024","unstructured":"Pengcheng Li, Xulong Zhang, Jing Xiao, and Jianzong Wang. 2024b. IDEAW: Robust Neural Audio Watermarking with Invertible Dual-Embedding. arXiv preprint arXiv:2409.19627 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Zero-shot fake video detection by audio-visual consistency. arXiv preprint arXiv:2406.07854","author":"Li Xiaolou","year":"2024","unstructured":"Xiaolou Li, Zehua Liu, Chen Chen, Lantian Li, Li Guo, and Dong Wang. 2024a. Zero-shot fake video detection by audio-visual consistency. arXiv preprint arXiv:2406.07854 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Exposing deepfake videos by detecting face warping artifacts. arXiv preprint arXiv:1811.00656","author":"Li Yuezun","year":"2018","unstructured":"Yuezun Li and Siwei Lyu. 2018. Exposing deepfake videos by detecting face warping artifacts. arXiv preprint arXiv:1811.00656 (2018)."},{"key":"e_1_3_2_1_24_1","volume-title":"Evolving from Single-modal to Multi-modal Facial Deepfake Detection: A Survey. arXiv preprint arXiv:2406.06965","author":"Liu Ping","year":"2024","unstructured":"Ping Liu, Qiqi Tao, and Joey Tianyi Zhou. 2024b. Evolving from Single-modal to Multi-modal Facial Deepfake Detection: A Survey. arXiv preprint arXiv:2406.06965 (2024)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2892"},{"key":"e_1_3_2_1_26_1","volume-title":"Dvmark: a deep multiscale framework for video watermarking","author":"Luo Xiyang","year":"2023","unstructured":"Xiyang Luo, Yinxiao Li, Huiwen Chang, Ce Liu, Peyman Milanfar, and Feng Yang. 2023. Dvmark: a deep multiscale framework for video watermarking. IEEE Transactions on Image Processing (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413570"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCNC51644.2023.10059841"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3640466"},{"key":"e_1_3_2_1_30_1","volume-title":"Passive Deepfake Detection Across Multi-modalities: A Comprehensive Survey. arXiv preprint arXiv:2411.17911","author":"Nguyen-Le Hong-Hanh","year":"2024","unstructured":"Hong-Hanh Nguyen-Le, Van-Tuan Tran, Dinh-Thuc Nguyen, and Nhien-An Le-Khac. 2024. Passive Deepfake Detection Across Multi-modalities: A Comprehensive Survey. arXiv preprint arXiv:2411.17911 (2024)."},{"key":"e_1_3_2_1_31_1","first-page":"27092","volume-title":"AVFF: Audio-Visual Feature Fusion for Video Deepfake Detection. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024","author":"Oorloff Trevine","year":"2024","unstructured":"Trevine Oorloff, Surya Koppisetti, Nicol\u00f2 Bonettini, Divyaraj Solanki, Ben Colman, Yaser Yacoob, Ali Shahriyari, and Gaurav Bharaj. 2024a. AVFF: Audio-Visual Feature Fusion for Video Deepfake Detection. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16-22, 2024. IEEE, 27092-27102."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02559"},{"key":"e_1_3_2_1_33_1","volume-title":"Munich","volume":"851","author":"Pumarola Albert","year":"2018","unstructured":"Albert Pumarola, Antonio Agudo, Aleix M. Mart\u00ednez, Alberto Sanfeliu, and Francesc Moreno-Noguer. 2018. GANimation: Anatomically-Aware Facial Animation from a Single Image. In Computer Vision - ECCV 2018 - 15th European Conference, Munich, Germany, September 8-14, 2018, Proceedings, Part X (Lecture Notes in Computer Science, Vol. 11214), Vittorio Ferrari, Martial Hebert, Cristian Sminchisescu, and Yair Weiss (Eds.). Springer, 835-851."},{"key":"e_1_3_2_1_34_1","volume-title":"2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). IEEE, 1786-1793","author":"Qureshi Amna","year":"2021","unstructured":"Amna Qureshi, David Meg\u00edas, and Minoru Kuribayashi. 2021. Detecting deepfake videos using digital watermarking. In 2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). IEEE, 1786-1793."},{"key":"e_1_3_2_1_35_1","first-page":"75","volume-title":"7th International Conference on Information and Computer Technologies, ICICT 2024","year":"2024","unstructured":"Md. Shohel Rana and Andrew H. Sung. 2024. Advanced Deepfake Detection using Machine Learning Algorithms: A Statistical Analysis and Performance Comparison. In 7th International Conference on Information and Computer Technologies, ICICT 2024, Honolulu, HI, USA, March 15-17, 2024. IEEE, 75-81."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2020.107652"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3629976"},{"key":"e_1_3_2_1_38_1","volume-title":"Mlp-mixer: An all-mlp architecture for vision. Advances in neural information processing systems","author":"Tolstikhin Ilya O","year":"2021","unstructured":"Ilya O Tolstikhin, Neil Houlsby, Alexander Kolesnikov, Lucas Beyer, Xiaohua Zhai, Thomas Unterthiner, Jessica Yung, Andreas Steiner, Daniel Keysers, Jakob Uszkoreit, et al., 2021. Mlp-mixer: An all-mlp architecture for vision. Advances in neural information processing systems, Vol. 34 (2021), 24261-24272."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26701"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680869"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2024.104133"},{"key":"e_1_3_2_1_42_1","volume-title":"Pitch Imperfect: Detecting Audio Deepfakes Through Acoustic Prosodic Analysis. arXiv preprint arXiv:2502.14726","author":"Warren Kevin","year":"2025","unstructured":"Kevin Warren, Daniel Olszewski, Seth Layton, Kevin Butler, Carrie Gates, and Patrick Traynor. 2025. Pitch Imperfect: Detecting Audio Deepfakes Through Acoustic Prosodic Analysis. arXiv preprint arXiv:2502.14726 (2025)."},{"key":"e_1_3_2_1_43_1","volume-title":"Robust Audio Watermarking Against Manipulation Attacks Based on Deep Learning","author":"Wen Shuangbing","year":"2024","unstructured":"Shuangbing Wen, Qishan Zhang, Tao Hu, and Jun Li. 2024. Robust Audio Watermarking Against Manipulation Attacks Based on Deep Learning. IEEE Signal Processing Letters (2024)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612471"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2020.3045937"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2021.3102487"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475628"},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 32nd ACM International Conference on Multimedia. 9818-9827","author":"Zhang Xuanyu","year":"2024","unstructured":"Xuanyu Zhang, Youmin Xu, Runyi Li, Jiwen Yu, Weiqi Li, Zhipei Xu, and Jian Zhang. 2024. V2a-mark: Versatile deep visual-audio watermarking for manipulation localization and copyright protection. In Proceedings of the 32nd ACM International Conference on Multimedia. 9818-9827."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612270"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01453"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755783","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:38:59Z","timestamp":1765309139000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755783"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":50,"alternative-id":["10.1145\/3746027.3755783","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755783","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}