{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:09:35Z","timestamp":1765008575529,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,9]]},"DOI":"10.1145\/3743093.3771003","type":"proceedings-article","created":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:06:16Z","timestamp":1765008376000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Frequency-Enhanced Multi-Modal Consistency Learning for Audio-Visual Deepfake Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8307-2821","authenticated-orcid":false,"given":"Xinyu","family":"Zheng","sequence":"first","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7218-8133","authenticated-orcid":false,"given":"Dongdong","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5686-5957","authenticated-orcid":false,"given":"Chengyu","family":"Sun","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Darius Afchar Vincent Nozick Junichi Yamagishi and Isao Echizen. 2018. MesoNet: a Compact Facial Video Forgery Detection Network.","DOI":"10.1109\/WIFS.2018.8630761"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"crossref","unstructured":"Simone Barattin Christos Tzelepis I. Patras and N. Sebe. 2023. Attribute-preserving Face Dataset Anonymization via Latent Code Optimization. ArXiv abs\/2303.11296 (2023).","DOI":"10.1109\/CVPR52729.2023.00773"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"crossref","unstructured":"Zhixi Cai Shreya Ghosh Aman\u00a0Pankaj Adatia Munawar Hayat Abhinav Dhall Tom Gedeon and Kalin Stefanov. 2023. AV-Deepfake1M: A Large-Scale LLM-Driven Audio-Visual Deepfake Dataset. (2023).","DOI":"10.1145\/3664647.3680795"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Zhixi Cai Kalin Stefanov Abhinav Dhall and Munawar Hayat. 2022. Do You Really Mean That? Content Driven Audio-Visual Deepfake Dataset and Multimodal Method for Temporal Forgery Localization.","DOI":"10.1109\/DICTA56598.2022.10034605"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00408"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Harry Cheng Yangyang Guo Tianyi Wang Qi Li Xiaojun Chang and Liqiang Nie. 2024. Voice-Face Homogeneity Tells Deepfake. ACM Transactions on Multimedia Computing Communications and Applications (TOMCCAP) 20 3 (2024) 22.","DOI":"10.1145\/3625231"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Komal Chugh Parul Gupta Abhinav Dhall and Ramanathan Subramanian. 2020. Not made for each other- Audio-Visual Dissonance-based Deepfake Detection and Localization. (2020).","DOI":"10.1145\/3394171.3413700"},{"key":"e_1_3_3_1_9_2","first-page":"1929","volume-title":"Interspeech","author":"Chung Joon\u00a0Son","year":"2018","unstructured":"Joon\u00a0Son Chung, Arsha Nagrani, and Andrew Zisserman. 2018. VoxCeleb2: Deep speaker recognition. In Interspeech. 1929\u20131933."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"crossref","unstructured":"Davide Cozzolino Matthias Niener and Luisa Verdoliva. 2022. Audio-Visual Person-of-Interest DeepFake Detection.","DOI":"10.1109\/CVPRW59228.2023.00101"},{"key":"e_1_3_3_1_11_2","unstructured":"Brian Dolhansky Joanna Bitton Ben Pflaum Jikuo Lu Russ Howes Menglin Wang and Cristian\u00a0Canton Ferrer. 2020. The deepfake detection challenge (dfdc) dataset."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01011"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2020. Generative adversarial networks. Commun. ACM 63 11 139\u2013144.","DOI":"10.1145\/3422622"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Alexandros Haliassos Rodrigo Mira Stavros Petridis and Maja Pantic. 2022. Leveraging real talking faces via self-supervision for robust forgery detection.","DOI":"10.1109\/CVPR52688.2022.01453"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"Alexandros Haliassos Konstantinos Vougioukas Stavros Petridis and Maja Pantic. 2021. Lips don\u2019t lie: A generalisable and robust approach to face forgery detection. 11438\u201311448\u00a0pages.","DOI":"10.1109\/CVPR46437.2021.00500"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2019.8803740"},{"key":"e_1_3_3_1_17_2","unstructured":"Shawn Hershey Sourish Chaudhuri Daniel P.\u00a0W. Ellis Jort\u00a0F. Gemmeke and Kevin Wilson. 2016. CNN Architectures for Large-Scale Audio Classification. IEEE (2016)."},{"key":"e_1_3_3_1_18_2","unstructured":"Tero Karras Timo Aila Samuli Laine and Jaakko Lehtinen. 2017. Progressive Growing of GANs for Improved Quality Stability and Variation. (2017)."},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Tero Karras Samuli Laine and Timo Aila. 2021. A Style-Based Generator Architecture for Generative Adversarial Networks. IEEE Transactions on Pattern Analysis and Machine Intelligence 43 12 (2021).","DOI":"10.1109\/TPAMI.2020.2970919"},{"key":"e_1_3_3_1_20_2","unstructured":"Hasam Khalid Shahroz Tariq and Simon\u00a0S. Woo. 2021. FakeAVCeleb: A Novel Audio-Video Multimodal Deepfake Dataset. (2021)."},{"key":"e_1_3_3_1_21_2","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1412.6980 (2014)."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Patrick Kwon Jaeseong You Gyuhyeon Nam Sungwoo Park and Gyeongsu Chae. 2021. KoDF: A Large-scale Korean DeepFake Detection Dataset. (2021).","DOI":"10.1109\/ICCV48922.2021.01057"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Trisha Mittal Uttaran Bhattacharya Rohan Chandra Aniket Bera and Dinesh Manocha. 2020. Emotions don\u2019t lie: An audio-visual deepfake detection method using affective cues. 2823\u20132832\u00a0pages.","DOI":"10.1145\/3394171.3413570"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02559"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58610-2_6"},{"key":"e_1_3_3_1_26_2","unstructured":"Alec Radford Luke Metz and Soumith Chintala. 2015. Unsupervised representation learning with deep convolutional generative adversarial networks."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00009"},{"key":"e_1_3_3_1_28_2","unstructured":"Bowen Shi Wei-Ning Hsu Kushal Lakhotia and Abdelrahman Mohamed. 2022. Learning audio-visual speech representation by masked multimodal cluster prediction."},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01816"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Chuangchuang Tan Yao Zhao Shikui Wei Guanghua Gu Ping Liu and Yunchao Wei. 2024. Frequency-aware deepfake detection: Improving generalizability through frequency space domain learning. 38 5 (2024) 5052\u20135060.","DOI":"10.1609\/aaai.v38i5.28310"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01165"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Du Tran Heng Wang Lorenzo Torresani Jamie Ray Yann LeCun and Manohar Paluri. 2018. A closer look at spatiotemporal convolutions for action recognition. 6450\u20136459\u00a0pages.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_3_1_33_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Rui Wang Dengpan Ye Long Tang Yunming Zhang and Jiacheng Deng. 2024. AVT2-DWF: Improving Deepfake Detection with Audio-Visual Fusion and Dynamic Weighting Strategies.","DOI":"10.1109\/LSP.2024.3433596"},{"key":"e_1_3_3_1_35_2","unstructured":"Deressa Wodajo and Solomon Atnafu. 2021. Deepfake video detection using convolutional vision transformer."},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"crossref","unstructured":"Wenyuan Yang Xiaoyu Zhou Zhikai Chen Bofei Guo Zhongjie Ba Zhihua Xia Xiaochun Cao and Kui Ren. 2023. Avoid-df: Audio-visual joint learning for detecting deepfake. IEEE Transactions on Information Forensics and Security 18 (2023) 2015\u20132029.","DOI":"10.1109\/TIFS.2023.3262148"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683164"},{"key":"e_1_3_3_1_38_2","unstructured":"Wenzhen Yue Yong Liu Xianghua Ying Bowei Xing Ruohao Guo and Ji Shi. 2025. Freeformer: Frequency enhanced transformer for multivariate time series forecasting."},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"crossref","unstructured":"Yibo Zhang Weiguo Lin and Junfeng Xu. 2024. Joint audio-visual attention with contrastive learning for more general deepfake detection. ACM Transactions on Multimedia Computing Communications and Applications 20 5 (2024) 1\u201323.","DOI":"10.1145\/3625100"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"crossref","unstructured":"Yibo Zhang Weiguo Lin and Junfeng Xu. 2024. Joint Audio-Visual Attention with Contrastive Learning for More General Deepfake Detection. ACM Transactions on Multimedia Computing Communications and Applications (TOMCCAP) 20 5 (2024) 23.","DOI":"10.1145\/3625100"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01453"}],"event":{"name":"MMAsia '25: ACM Multimedia Asia","location":"Kuala Lumpur Malaysia","acronym":"MMAsia '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 7th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3743093.3771003","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:06:39Z","timestamp":1765008399000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3743093.3771003"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":40,"alternative-id":["10.1145\/3743093.3771003","10.1145\/3743093"],"URL":"https:\/\/doi.org\/10.1145\/3743093.3771003","relation":{},"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"2025-12-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}