{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T15:04:35Z","timestamp":1777043075049,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":73,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,18]]},"DOI":"10.1145\/3733102.3733133","type":"proceedings-article","created":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T11:14:07Z","timestamp":1750158847000},"page":"12-23","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Modality-Agnostic Deepfakes Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0659-0967","authenticated-orcid":false,"given":"Yu","family":"Cai","sequence":"first","affiliation":[{"name":"University at Buffalo;Institute of Information Engineering, Chinese Academy of Sciences;School of Cyber Security, University of Chinese Academy of Sciences, Buffalo, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6010-2963","authenticated-orcid":false,"given":"Peng","family":"Chen","sequence":"additional","affiliation":[{"name":"RealAI Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7863-3551","authenticated-orcid":false,"given":"Jiahe","family":"Tian","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences;School of Cyber Security, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9106-8630","authenticated-orcid":false,"given":"Jin","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences;School of Cyber Security, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3559-8009","authenticated-orcid":false,"given":"Jiao","family":"Dai","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6642-8160","authenticated-orcid":false,"given":"Xi","family":"Wang","sequence":"additional","affiliation":[{"name":"Institute of Microelectronics, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7503-8378","authenticated-orcid":false,"given":"Shan","family":"Jia","sequence":"additional","affiliation":[{"name":"University at Buffalo, Buffalo, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0992-685X","authenticated-orcid":false,"given":"Siwei","family":"Lyu","sequence":"additional","affiliation":[{"name":"University at Buffalo, Buffalo, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1107-3873","authenticated-orcid":false,"given":"Jizhong","family":"Han","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,17]]},"reference":[{"key":"e_1_3_3_2_2_2","first-page":"1","volume-title":"WIFS","author":"Afchar Darius","year":"2018","unstructured":"Darius Afchar, Vincent Nozick, Junichi Yamagishi, and Isao Echizen. 2018. Mesonet: a compact facial video forgery detection network. In WIFS. IEEE, 1\u20137."},{"key":"e_1_3_3_2_3_2","unstructured":"Triantafyllos Afouras Joon\u00a0Son Chung Andrew Senior Oriol Vinyals and Andrew Zisserman. 2018. Deep audio-visual speech recognition. TPAMI (2018)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00338"},{"key":"e_1_3_3_2_5_2","first-page":"38","volume-title":"CVPR Workshops","author":"Agarwal Shruti","year":"2019","unstructured":"Shruti Agarwal, Hany Farid, Yuming Gu, Mingming He, Koki Nagano, and Hao Li. 2019. Protecting World Leaders Against Deep Fakes.. In CVPR Workshops, Vol.\u00a01. 38."},{"key":"e_1_3_3_2_6_2","first-page":"17","volume-title":"AVSP","author":"Almajai Ibrahim","year":"2007","unstructured":"Ibrahim Almajai and Ben Milner. 2007. Maximising audio-visual speech correlation.. In AVSP. 17."},{"key":"e_1_3_3_2_7_2","unstructured":"Alexei Baevski Yuhao Zhou Abdelrahman Mohamed and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. NeurIPS 33 (2020) 12449\u201312460."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Zhixi Cai Kalin Stefanov Abhinav Dhall and Munawar Hayat. 2022. Do You Really Mean That? Content Driven Audio-Visual Deepfake Dataset and Multimodal Method for Temporal Forgery Localization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06228 (2022).","DOI":"10.1109\/DICTA56598.2022.10034605"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Ruth Campbell. 2008. The processing of audio-visual speech: empirical and neural bases. Philosophical Transactions of the Royal Society B: Biological Sciences 363 1493 (2008) 1001\u20131010.","DOI":"10.1098\/rstb.2007.2155"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01815"},{"key":"e_1_3_3_2_11_2","unstructured":"Harry Cheng Yangyang Guo Tianyi Wang Qi Li Tao Ye and Liqiang Nie. 2022. Voice-Face Homogeneity Tells Deepfake. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2203.02195 (2022)."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"crossref","unstructured":"Kun Cheng Xiaodong Cun Yong Zhang Menghan Xia Fei Yin Mingrui Zhu Xuan Wang Jue Wang and Nannan Wang. 2022. VideoReTalking: Audio-based Lip Synchronization for Talking Head Video Editing In the Wild. arxiv:https:\/\/arXiv.org\/abs\/2211.14758\u00a0[cs.CV]","DOI":"10.1145\/3550469.3555399"},{"key":"e_1_3_3_2_13_2","unstructured":"Xing Cheng Hezheng Lin Xiangyu Wu Fan Yang Dong Shen Zhongyuan Wang Nian Shi and Honglin Liu. 2021. MlTr: Multi-label Classification with Transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.06195 (2021)."},{"key":"e_1_3_3_2_14_2","first-page":"439","volume-title":"MM","author":"Chugh Komal","year":"2020","unstructured":"Komal Chugh, Parul Gupta, Abhinav Dhall, and Ramanathan Subramanian. 2020. Not made for each other-audio-visual dissonance-based deepfake detection and localization. In MM. 439\u2013447."},{"key":"e_1_3_3_2_15_2","first-page":"2977","volume-title":"Interspeech","author":"Chung Joon\u00a0Son","year":"2020","unstructured":"Joon\u00a0Son Chung, Jaesung Huh, Seongkyu Mun, Minjae Lee, Hee-Soo Heo, Soyeon Choe, Chiheon Ham, Sung-Ye Jung, Bong-Jin Lee, and Icksang Han. 2020. In defence of metric learning for speaker recognition. In Interspeech. 2977\u20132981."},{"key":"e_1_3_3_2_16_2","first-page":"219","volume-title":"ICIP","author":"Coccomini Davide\u00a0Alessandro","year":"2022","unstructured":"Davide\u00a0Alessandro Coccomini, Nicola Messina, Claudio Gennaro, and Fabrizio Falchi. 2022. Combining efficientnet and vision transformers for video deepfake detection. In ICIP. Springer, 219\u2013229."},{"key":"e_1_3_3_2_17_2","first-page":"943","volume-title":"CVPR Workshops","author":"Cozzolino Davide","year":"2023","unstructured":"Davide Cozzolino, Alessandro Pianese, Matthias Nie\u00dfner, and Luisa Verdoliva. 2023. Audio-Visual Person-of-Interest DeepFake Detection. In CVPR Workshops. 943\u2013952."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Brecht Desplanques Jenthe Thienpondt and Kris Demuynck. 2020. Ecapa-tdnn: Emphasized channel attention propagation and aggregation in tdnn based speaker verification. (2020) 3830\u20133834.","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"e_1_3_3_2_19_2","unstructured":"Brian Dolhansky Russ Howes Ben Pflaum Nicole Baram and Cristian\u00a0Canton Ferrer. 2019. The deepfake detection challenge (dfdc) preview dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.08854 (2019)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Chao Feng Ziyang Chen and Andrew Owens. 2023. Self-Supervised Video Forensics by Audio-Visual Anomaly Detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.01767 (2023).","DOI":"10.1109\/CVPR52729.2023.01011"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_3_2_22_2","first-page":"735","volume-title":"AAAI","author":"Gu Qiqi","year":"2022","unstructured":"Qiqi Gu, Shen Chen, Taiping Yao, Yang Chen, Shouhong Ding, and Ran Yi. 2022. Exploiting Fine-grained Face Forgery Clues via Progressive Enhancement Learning. In AAAI, Vol.\u00a036. 735\u2013743."},{"key":"e_1_3_3_2_23_2","first-page":"168","volume-title":"IWDW","author":"Gu Yewei","year":"2020","unstructured":"Yewei Gu, Xianfeng Zhao, Chen Gong, and Xiaowei Yi. 2020. Deepfake Video Detection Using Audio-Visual Consistency. In IWDW. Springer, 168\u2013180."},{"key":"e_1_3_3_2_24_2","first-page":"3473","volume-title":"CVPR","author":"Gu Zhihao","year":"2021","unstructured":"Zhihao Gu, Yang Chen, Taiping Yao, Shouhong Ding, Jilin Li, Feiyue Huang, and Lizhuang Ma. 2021. Spatiotemporal inconsistency learning for deepfake video detection. In CVPR. 3473\u20133481."},{"key":"e_1_3_3_2_25_2","first-page":"744","volume-title":"AAAI","author":"Gu Zhihao","year":"2022","unstructured":"Zhihao Gu, Yang Chen, Taiping Yao, Shouhong Ding, Jilin Li, and Lizhuang Ma. 2022. Delving into the Local: Dynamic Inconsistency Learning for DeepFake Video Detection. In AAAI. 744\u2013752."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01453"},{"key":"e_1_3_3_2_27_2","first-page":"5039","volume-title":"CVPR","author":"Haliassos Alexandros","year":"2021","unstructured":"Alexandros Haliassos, Konstantinos Vougioukas, Stavros Petridis, and Maja Pantic. 2021. Lips don\u2019t lie: A generalisable and robust approach to face forgery detection. In CVPR. 5039\u20135049."},{"key":"e_1_3_3_2_28_2","unstructured":"Hee\u00a0Soo Heo Bong-Jin Lee Jaesung Huh and Joon\u00a0Son Chung. 2020. Clova baseline system for the voxceleb speaker recognition challenge 2020. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2009.14153 (2020)."},{"key":"e_1_3_3_2_29_2","first-page":"1013","volume-title":"CVPR","author":"Hosler Brian","year":"2021","unstructured":"Brian Hosler, Davide Salvi, Anthony Murray, Fabio Antonacci, Paolo Bestagini, Stefano Tubaro, and Matthew\u00a0C Stamm. 2021. Do deepfakes feel emotions? A semantic approach to detecting deepfakes via emotional inconsistencies. In CVPR. 1013\u20131022."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"crossref","unstructured":"ICME. 2021. Defakehop: A light-weight high-performance deepfake detector. IEEE 1\u20136.","DOI":"10.1109\/ICME51207.2021.9428361"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Hafsa Ilyas Ali Javed and Khalid\u00a0Mahmood Malik. 2023. AVFakeNet: A unified end-to-end Dense Swin Transformer deep learning model for audio\u2013visual deepfakes detection. Applied Soft Computing 136 (2023) 110124.","DOI":"10.1016\/j.asoc.2023.110124"},{"key":"e_1_3_3_2_32_2","first-page":"18","volume-title":"SLT Workshop","author":"Kameoka H","year":"2018","unstructured":"H Kameoka, T Kaneko, K Tanaka, and N\u00a0StarGAN-VC Hojo. 2018. Non-parallel many-to-many voice conversion using star generative adversarial networks. In SLT Workshop. 18\u201321."},{"key":"e_1_3_3_2_33_2","volume-title":"NeurIPS","author":"Khalid Hasam","year":"2021","unstructured":"Hasam Khalid, Shahroz Tariq, Minha Kim, and Simon\u00a0S. Woo. 2021. FakeAVCeleb: A Novel Audio-Video Multimodal Deepfake Dataset. In NeurIPS."},{"key":"e_1_3_3_2_34_2","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1412.6980 (2014)."},{"key":"e_1_3_3_2_35_2","first-page":"3677","volume-title":"ICCV","author":"Korshunova Iryna","year":"2017","unstructured":"Iryna Korshunova, Wenzhe Shi, Joni Dambre, and Lucas Theis. 2017. Fast face-swap using convolutional neural networks. In ICCV. 3677\u20133685."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00505"},{"key":"e_1_3_3_2_37_2","first-page":"1","volume-title":"WIFS","author":"Li Yuezun","year":"2018","unstructured":"Yuezun Li, Ming-Ching Chang, and Siwei Lyu. 2018. In ictu oculi: Exposing ai created fake videos by detecting eye blinking. In WIFS. IEEE, 1\u20137."},{"key":"e_1_3_3_2_38_2","first-page":"2302","volume-title":"AAAI","author":"Ma Mengmeng","year":"2021","unstructured":"Mengmeng Ma, Jian Ren, Long Zhao, Sergey Tulyakov, Cathy Wu, and Xi Peng. 2021. SMIL: Multimodal learning with severely missing modality. In AAAI, Vol.\u00a035. 2302\u20132310."},{"key":"e_1_3_3_2_39_2","first-page":"7613","volume-title":"ICASSP","author":"Ma Pingchuan","year":"2021","unstructured":"Pingchuan Ma, Stavros Petridis, and Maja Pantic. 2021. End-to-end audio-visual speech recognition with conformers. In ICASSP. IEEE, 7613\u20137617."},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9004036"},{"key":"e_1_3_3_2_41_2","first-page":"9241","volume-title":"ICASSP","author":"Mart\u00edn-Do\u00f1as Juan\u00a0M","year":"2022","unstructured":"Juan\u00a0M Mart\u00edn-Do\u00f1as and Aitor \u00c1lvarez. 2022. The Vicomtech Audio Deepfake Detection System Based on Wav2vec2 for the 2022 ADD Challenge. In ICASSP. IEEE, 9241\u20139245."},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW.2019.00020"},{"key":"e_1_3_3_2_43_2","first-page":"2823","volume-title":"MM","author":"Mittal Trisha","year":"2020","unstructured":"Trisha Mittal, Uttaran Bhattacharya, Rohan Chandra, Aniket Bera, and Dinesh Manocha. 2020. Emotions don\u2019t lie: An audio-visual deepfake detection method using affective cues. In MM. 2823\u20132832."},{"key":"e_1_3_3_2_44_2","first-page":"2307","volume-title":"ICASSP","author":"Nguyen Huy\u00a0H","year":"2019","unstructured":"Huy\u00a0H Nguyen, Junichi Yamagishi, and Isao Echizen. 2019. Capsule-forensics: Using capsule networks to detect forged images and videos. In ICASSP. IEEE, 2307\u20132311."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952625"},{"key":"e_1_3_3_2_46_2","first-page":"1","volume-title":"WIFS","author":"Pianese Alessandro","year":"2022","unstructured":"Alessandro Pianese, Davide Cozzolino, Giovanni Poggi, and Luisa Verdoliva. 2022. Deepfake audio detection by speaker verification. In WIFS. IEEE, 1\u20136."},{"key":"e_1_3_3_2_47_2","unstructured":"Wei Ping Kainan Peng and Jitong Chen. 2018. Clarinet: Parallel wave generation in end-to-end text-to-speech. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1807.07281 (2018)."},{"key":"e_1_3_3_2_48_2","first-page":"484","volume-title":"MM","author":"Prajwal KR","year":"2020","unstructured":"KR Prajwal, Rudrabha Mukhopadhyay, Vinay\u00a0P Namboodiri, and CV Jawahar. 2020. A lip sync expert is all you need for speech to lip generation in the wild. In MM. 484\u2013492."},{"key":"e_1_3_3_2_49_2","first-page":"1","volume-title":"ICCV","author":"Rossler Andreas","year":"2019","unstructured":"Andreas Rossler, Davide Cozzolino, Luisa Verdoliva, Christian Riess, Justus Thies, and Matthias Nie\u00dfner. 2019. Faceforensics++: Learning to detect manipulated facial images. In ICCV. 1\u201311."},{"key":"e_1_3_3_2_50_2","volume-title":"ICLR","author":"Shi Bowen","year":"2021","unstructured":"Bowen Shi, Wei-Ning Hsu, Kushal Lakhotia, and Abdelrahman Mohamed. 2021. Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction. In ICLR."},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01816"},{"key":"e_1_3_3_2_52_2","unstructured":"Aliaksandr Siarohin St\u00e9phane Lathuili\u00e8re Sergey Tulyakov Elisa Ricci and Nicu Sebe. 2019. First order motion model for image animation. NeurIPS 32 (2019)."},{"key":"e_1_3_3_2_53_2","first-page":"6447","volume-title":"CVPR","author":"Son\u00a0Chung Joon","year":"2017","unstructured":"Joon Son\u00a0Chung, Andrew Senior, Oriol Vinyals, and Andrew Zisserman. 2017. Lip reading sentences in the wild. In CVPR. 6447\u20136456."},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"crossref","unstructured":"Themos Stafylakis and Georgios Tzimiropoulos. 2017. Combining Residual Networks with LSTMs for Lipreading. Interspeech (2017) 3652\u20133656.","DOI":"10.21437\/Interspeech.2017-85"},{"key":"e_1_3_3_2_55_2","first-page":"3609","volume-title":"CVPR","author":"Sun Zekun","year":"2021","unstructured":"Zekun Sun, Yujie Han, Zeyu Hua, Na Ruan, and Weijia Jia. 2021. Improving the efficiency and robustness of deepfakes detection through precise geometric features. In CVPR. 3609\u20133618."},{"key":"e_1_3_3_2_56_2","unstructured":"Hemlata Tak Jee-weon Jung Jose Patino Madhu Kamble Massimiliano Todisco and Nicholas Evans. 2021. End-to-end spectro-temporal graph attention networks for speaker verification anti-spoofing and speech deepfake detection. Automatic Speaker Verification and Spoofing Countermeasures Challenge (2021)."},{"key":"e_1_3_3_2_57_2","first-page":"6369","volume-title":"ICASSP","author":"Tak Hemlata","year":"2021","unstructured":"Hemlata Tak, Jose Patino, Massimiliano Todisco, Andreas Nautsch, Nicholas Evans, and Anthony Larcher. 2021. End-to-end anti-spoofing with rawnet2. In ICASSP. IEEE, 6369\u20136373."},{"key":"e_1_3_3_2_58_2","first-page":"2387","volume-title":"CVPR","author":"Thies Justus","year":"2016","unstructured":"Justus Thies, Michael Zollhofer, Marc Stamminger, Christian Theobalt, and Matthias Nie\u00dfner. 2016. Face2face: Real-time face capture and reenactment of rgb videos. In CVPR. 2387\u20132395."},{"key":"e_1_3_3_2_59_2","unstructured":"Laurens Van\u00a0der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. JMLR 9 11 (2008)."},{"key":"e_1_3_3_2_60_2","first-page":"5998","volume-title":"NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In NeurIPS. 5998\u20136008."},{"key":"e_1_3_3_2_61_2","first-page":"4259","volume-title":"Interspeech","author":"Wang Xin","year":"2021","unstructured":"Xin Wang and Junichi Yamagishi. 2021. A Comparative Study on Recent Neural Spoofing Countermeasures for Synthetic Speech Detection. In Interspeech. 4259\u20134263."},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"crossref","unstructured":"Xin Wang and Junichi Yamagishi. 2021. Investigating self-supervised front ends for speech spoofing countermeasures. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.07725 (2021).","DOI":"10.21437\/Odyssey.2022-14"},{"key":"e_1_3_3_2_63_2","unstructured":"Deressa Wodajo and Solomon Atnafu. 2021. Deepfake video detection using convolutional vision transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2102.11126 (2021)."},{"key":"e_1_3_3_2_64_2","first-page":"4269","volume-title":"Interspeech","author":"Xie Yang","year":"2021","unstructured":"Yang Xie, Zhenchuan Zhang, and Yingchun Yang. 2021. Siamese Network with wav2vec Feature for Spoofing Speech Detection.. In Interspeech. 4269\u20134273."},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"crossref","unstructured":"Wenyuan Yang Xiaoyu Zhou Zhikai Chen Bofei Guo Zhongjie Ba Zhihua Xia Xiaochun Cao and Kui Ren. 2023. AVoiD-DF: Audio-Visual Joint Learning for Detecting Deepfake. TIFS 18 (2023) 2015\u20132029.","DOI":"10.1109\/TIFS.2023.3262148"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683164"},{"key":"e_1_3_3_2_67_2","first-page":"1","volume-title":"ICME","author":"Yu Cai","year":"2022","unstructured":"Cai Yu, Peng Chen, Jiao Dai, Xi Wang, Weibo Zhang, Jin Liu, and Jizhong Han. 2022. Focus by Prior: Deepfake Detection Based on Prior-Attention. In ICME. IEEE, 1\u20136."},{"key":"e_1_3_3_2_68_2","volume-title":"ICME","author":"Yu Cai","year":"2024","unstructured":"Cai Yu, Shan Jia, Xiaomeng Fu, Jin Liu, Jiahe Tian, Jiao Dai, Xi Wang, Siwei Lyu, and Jizhong Han. 2024. Explicit Correlation Learning for Generalizable Cross-Modal Deepfake Detection. In ICME. IEEE."},{"key":"e_1_3_3_2_69_2","unstructured":"Yang Yu Xiaolong Liu Rongrong Ni Siyuan Yang Yao Zhao and Alex\u00a0C Kot. 2023. PVASS-MDD: Predictive Visual-audio Alignment Self-supervision for Multimodal Deepfake Detection. TCSVT (2023)."},{"key":"e_1_3_3_2_70_2","first-page":"1288","volume-title":"IJCAI","author":"Zhang Daichi","year":"2021","unstructured":"Daichi Zhang, Chenyu Li, Fanzhao Lin, Dan Zeng, and Shiming Ge. 2021. Detecting Deepfake Videos with Temporal Dropout 3DCNN.. In IJCAI. 1288\u20131294."},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682566"},{"key":"e_1_3_3_2_72_2","first-page":"192","volume-title":"ICCV","author":"Zhang Shifeng","year":"2017","unstructured":"Shifeng Zhang, Xiangyu Zhu, Zhen Lei, Hailin Shi, Xiaobo Wang, and Stan\u00a0Z Li. 2017. S3fd: Single shot scale-invariant face detector. In ICCV. 192\u2013201."},{"key":"e_1_3_3_2_73_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00222"},{"key":"e_1_3_3_2_74_2","first-page":"14800","volume-title":"ICCV","author":"Zhou Yipin","year":"2021","unstructured":"Yipin Zhou and Ser-Nam Lim. 2021. Joint audio-visual deepfake detection. In ICCV. 14800\u201314809."}],"event":{"name":"IH&MMSEC '25: ACM Workshop on Information Hiding and Multimedia Security","location":"San Jose USA","acronym":"IH&MMSEC '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the ACM Workshop on Information Hiding and Multimedia Security"],"original-title":[],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T11:16:14Z","timestamp":1750158974000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3733102.3733133"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,17]]},"references-count":73,"alternative-id":["10.1145\/3733102.3733133","10.1145\/3733102"],"URL":"https:\/\/doi.org\/10.1145\/3733102.3733133","relation":{},"subject":[],"published":{"date-parts":[[2025,6,17]]},"assertion":[{"value":"2025-06-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}