{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T15:42:41Z","timestamp":1780501361478,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3689145","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"11355-11359","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["1M-Deepfakes Detection Challenge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7978-0860","authenticated-orcid":false,"given":"Zhixi","family":"Cai","sequence":"first","affiliation":[{"name":"Monash University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2230-1440","authenticated-orcid":false,"given":"Abhinav","family":"Dhall","sequence":"additional","affiliation":[{"name":"Flinders University, Adelaide, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2639-8374","authenticated-orcid":false,"given":"Shreya","family":"Ghosh","sequence":"additional","affiliation":[{"name":"Curtin University, Perth, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2706-5985","authenticated-orcid":false,"given":"Munawar","family":"Hayat","sequence":"additional","affiliation":[{"name":"Qualcomm, San Diego, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8188-3751","authenticated-orcid":false,"given":"Dimitrios","family":"Kollias","sequence":"additional","affiliation":[{"name":"Queen Mary University of London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0861-8660","authenticated-orcid":false,"given":"Kalin","family":"Stefanov","sequence":"additional","affiliation":[{"name":"Monash University, Melbourne, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8244-2165","authenticated-orcid":false,"given":"Usman","family":"Tariq","sequence":"additional","affiliation":[{"name":"American University of Sharjah, Sharjah, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WIFS.2018.8630761"},{"key":"e_1_3_2_1_2_1","volume-title":"You Should Worry. Forbes (Oct.","author":"Brandon John","year":"2019","unstructured":"John Brandon. 2019. There Are Now 15,000 Deepfake Videos on Social Media. Yes, You Should Worry. Forbes (Oct. 2019)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_1_4_1","volume-title":"Munawar Hayat, Abhinav Dhall, and Kalin Stefanov.","author":"Cai Zhixi","year":"2023","unstructured":"Zhixi Cai, Shreya Ghosh, Aman Pankaj Adatia, Munawar Hayat, Abhinav Dhall, and Kalin Stefanov. 2023. AV-Deepfake1M: A Large-Scale LLM-Driven Audio-Visual Deepfake Dataset. arXiv:2311.15308 [cs]."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103818"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00150"},{"key":"e_1_3_2_1_7_1","volume-title":"Content Driven Audio-Visual Deepfake Dataset and Multimodal Method for Temporal Forgery Localization. In 2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA)","author":"Cai Zhixi","year":"2022","unstructured":"Zhixi Cai, Kalin Stefanov, Abhinav Dhall, and Munawar Hayat. 2022. Do You Really Mean That? Content Driven Audio-Visual Deepfake Dataset and Multimodal Method for Temporal Forgery Localization. In 2022 International Conference on Digital Image Computing: Techniques and Applications (DICTA). Sydney, Australia, 1--10."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413700"},{"key":"e_1_3_2_1_10_1","volume-title":"Image Analysis and Processing -- ICIAP 2022 (Lecture Notes in Computer Science), Stan Sclaroff, Cosimo Distante, Marco Leo, Giovanni M","author":"Coccomini Davide Alessandro","unstructured":"Davide Alessandro Coccomini, Nicola Messina, Claudio Gennaro, and Fabrizio Falchi. 2022. Combining EfficientNet and\u00a0Vision Transformers for\u00a0Video Deepfake Detection. In Image Analysis and Processing -- ICIAP 2022 (Lecture Notes in Computer Science), Stan Sclaroff, Cosimo Distante, Marco Leo, Giovanni M. Farinella, and Federico Tombari (Eds.). Springer International Publishing, Cham, 219--229."},{"key":"e_1_3_2_1_11_1","volume-title":"The DeepFake Detection Challenge (DFDC) Dataset. arXiv","author":"Dolhansky Brian","year":"2006","unstructured":"Brian Dolhansky, Joanna Bitton, Ben Pflaum, Jikuo Lu, Russ Howes, Menglin Wang, and Cristian Canton Ferrer. 2020. The DeepFake Detection Challenge (DFDC) Dataset. arXiv: 2006.07397 [cs]."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01011"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV) (Lecture Notes in Computer Science), Shai Avidan, Gabriel Brostow, Moustapha Ciss\u00e9","author":"Ge Songwei","unstructured":"Songwei Ge, Thomas Hayes, Harry Yang, Xi Yin, Guan Pang, David Jacobs, Jia-Bin Huang, and Devi Parikh. 2022. Long Video Generation with\u00a0Time-Agnostic VQGAN and\u00a0Time-Sensitive Transformer. In Proceedings of the European Conference on Computer Vision (ECCV) (Lecture Notes in Computer Science), Shai Avidan, Gabriel Brostow, Moustapha Ciss\u00e9, Giovanni Maria Farinella, and Tal Hassner (Eds.). Springer Nature Switzerland, Cham, 102--118."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00500"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00434"},{"key":"e_1_3_2_1_16_1","volume-title":"Applied Soft Computing","volume":"136","author":"Ilyas Hafsa","year":"2023","unstructured":"Hafsa Ilyas, Ali Javed, and Khalid Mahmood Malik. 2023. AVFakeNet: A unified end-to-end Dense Swin Transformer deep learning model for audio--visual deepfakes detection. Applied Soft Computing, Vol. 136 (March 2023), 110124."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00296"},{"key":"e_1_3_2_1_18_1","unstructured":"Ziyue Jiang Jinglin Liu Yi Ren Jinzheng He Chen Zhang Zhenhui Ye Pengfei Wei Chunfeng Wang Xiang Yin Zejun Ma and Zhou Zhao. 2023. Mega-TTS 2: Zero-Shot Text-to-Speech with Arbitrary Length Speech Prompts. arXiv:2307.07218 [cs eess]."},{"key":"e_1_3_2_1_19_1","unstructured":"Ziyue Jiang Yi Ren Zhenhui Ye Jinglin Liu Chen Zhang Qian Yang Shengpeng Ji Rongjie Huang Chunfeng Wang Xiang Yin Zejun Ma and Zhou Zhao. 2023. Mega-TTS: Zero-Shot Text-to-Speech at Scale with Intrinsic Inductive Bias. arXiv:2306.03509 [cs eess]."},{"key":"e_1_3_2_1_20_1","volume-title":"Woo","author":"Khalid Hasam","year":"2021","unstructured":"Hasam Khalid, Shahroz Tariq, and Simon S. Woo. 2021. FakeAVCeleb: A Novel Audio-Video Multimodal Deepfake Dataset. arXiv: 2108.05080 [cs]."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610661.3617148"},{"key":"e_1_3_2_1_22_1","unstructured":"Pavel Korshunov and Sebastien Marcel. 2018. DeepFakes: a New Threat to Face Recognition? Assessment and Detection. arXiv:1812.08685 [cs]."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01057"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00996"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00505"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00327"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3285283"},{"key":"e_1_3_2_1_28_1","volume-title":"Usman Tariq, Abhinav Dhall, Carlos Ivan Colon, and Hasan Al-Nashash.","author":"Naeem Shahzeb","year":"2024","unstructured":"Shahzeb Naeem, Muhammad Riyyan Khan, Usman Tariq, Abhinav Dhall, Carlos Ivan Colon, and Hasan Al-Nashash. 2024. Generation and Detection of Sign Language Deepfakes-A Linguistic and Visual Analysis. arXiv preprint arXiv:2404.01438 (2024)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00939"},{"key":"e_1_3_2_1_30_1","unstructured":"Dufou Nick and Jigsaw Andrew. 2019. Contributing Data to Deepfake Detection Research."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02559"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-205"},{"key":"e_1_3_2_1_33_1","volume-title":"Vigo: Audiovisual Fake Detection and Segment Localization. In ACM international conference on multimedia.","author":"P\u00e9rez-Vieites Diego","year":"2024","unstructured":"Diego P\u00e9rez-Vieites, Juan Jos\u00e9 Moreira-P\u00e9rez, \u00c1ngel Arag\u00f3n-Kifute, Raquel Rom\u00e1n-Sarmiento, and Rub\u00e9n Castro-Gonz\u00e1lez. 2024. Vigo: Audiovisual Fake Detection and Segment Localization. In ACM international conference on multimedia."},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 993--1000","author":"Raza Muhammad Anas","year":"2023","unstructured":"Muhammad Anas Raza and Khalid Mahmood Malik. 2023. Multimodaltrace: Deepfake Detection Using Audiovisual Representation Learning. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 993--1000."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00009"},{"key":"e_1_3_2_1_36_1","volume-title":"What are deepfakes --\u00a0and how can you spot them? The Guardian (Jan","author":"Sample Ian","year":"2020","unstructured":"Ian Sample. 2020. What are deepfakes --\u00a0and how can you spot them? The Guardian (Jan. 2020)."},{"key":"e_1_3_2_1_37_1","unstructured":"Conrad Sanderson (Ed.). 2002. The VidTIMIT Database. IDIAP."},{"key":"e_1_3_2_1_38_1","volume-title":"You thought fake news was bad? Deep fakes are where truth goes to die. The Guardian (Nov","author":"Schwartz Oscar","year":"2018","unstructured":"Oscar Schwartz. 2018. You thought fake news was bad? Deep fakes are where truth goes to die. The Guardian (Nov. 2018)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3367749"},{"key":"e_1_3_2_1_40_1","unstructured":"Kai Shen Zeqian Ju Xu Tan Yanqing Liu Yichong Leng Lei He Tao Qin Sheng Zhao and Jiang Bian. 2023. NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers. arXiv:2304.09116 [cs eess]."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01808"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01816"},{"key":"e_1_3_2_1_43_1","unstructured":"Uriel Singer Adam Polyak Thomas Hayes Xi Yin Jie An Songyang Zhang Qiyuan Hu Harry Yang Oron Ashual Oran Gafni Devi Parikh Sonal Gupta and Yaniv Taigman. 2022. Make-A-Video: Text-to-Video Generation without Text-Video Data. arXiv:2209.14792 [cs]."},{"key":"e_1_3_2_1_44_1","volume-title":"Deepfakes: A threat to democracy or just a bit of fun? BBC News (Jan.","author":"Thomas Daniel","year":"2020","unstructured":"Daniel Thomas. 2020. Deepfakes: A threat to democracy or just a bit of fun? BBC News (Jan. 2020)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512527.3531415"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"e_1_3_2_1_47_1","unstructured":"Yi Wang Kunchang Li Yizhuo Li Yinan He Bingkun Huang Zhiyu Zhao Hongjie Zhang Jilan Xu Yi Liu Zun Wang Sen Xing Guo Chen Junting Pan Jiashuo Yu Yali Wang Limin Wang and Yu Qiao. 2022. InternVideo: General Video Foundation Models via Generative and Discriminative Learning. arXiv:2212.03191 [cs]."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"e_1_3_2_1_49_1","volume-title":"Large-Scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation","author":"Wu Yusong","unstructured":"Yusong Wu, Ke Chen, Tianyu Zhang, Yuchen Hui, Taylor Berg-Kirkpatrick, and Shlomo Dubnov. 2023. Large-Scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 1--5. ISSN: 2379--190X."},{"key":"e_1_3_2_1_50_1","volume-title":"Patterns","volume":"3","author":"Xu Zhen","year":"2022","unstructured":"Zhen Xu, Sergio Escalera, Adrien Pavao, Magali Richard, Wei-Wei Tu, Quanming Yao, Huan Zhao, and Isabelle Guyon. 2022. Codabench: Flexible, easy-to-use, and reproducible meta-benchmark platform. Patterns, Vol. 3, 7 (2022)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12234"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2023.3262148"},{"key":"e_1_3_2_1_53_1","volume-title":"ADD 2022: the First Audio Deep Synthesis Detection Challenge. arXiv:2202","author":"Yi Jiangyan","year":"2022","unstructured":"Jiangyan Yi, Ruibo Fu, Jianhua Tao, Shuai Nie, Haoxin Ma, Chenglong Wang, Tao Wang, Zhengkun Tian, Ye Bai, Cunhang Fan, Shan Liang, Shiming Wang, Shuai Zhang, Xinrui Yan, Le Xu, Zhengqi Wen, Haizhou Li, Zheng Lian, and Bin Liu. 2022. ADD 2022: the First Audio Deep Synthesis Detection Challenge. arXiv:2202.08433 [cs, eess]."},{"key":"e_1_3_2_1_54_1","volume-title":"Kot","author":"Yu Yang","year":"2023","unstructured":"Yang Yu, Xiaolong Liu, Rongrong Ni, Siyuan Yang, Yao Zhao, and Alex C. Kot. 2023. PVASS-MDD: Predictive Visual-audio Alignment Self-supervision for Multimodal Deepfake Detection. IEEE Transactions on Circuits and Systems for Video Technology (2023), 1--1."},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV) (Lecture Notes in Computer Science), Shai Avidan, Gabriel Brostow, Moustapha Ciss\u00e9","author":"Zhang Chen-Lin","unstructured":"Chen-Lin Zhang, Jianxin Wu, and Yin Li. 2022. ActionFormer: Localizing Moments of\u00a0Actions with\u00a0Transformers. In Proceedings of the European Conference on Computer Vision (ECCV) (Lecture Notes in Computer Science), Shai Avidan, Gabriel Brostow, Moustapha Ciss\u00e9, Giovanni Maria Farinella, and Tal Hassner (Eds.). Springer Nature Switzerland, Cham, 492--510."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","unstructured":"Hang Zhang Xin Li and Lidong Bing. 2023. Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding. arXiv:2306.02858 [cs eess].","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613767"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00572"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413769"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3689145","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3689145","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:29Z","timestamp":1750295849000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3689145"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":59,"alternative-id":["10.1145\/3664647.3689145","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3689145","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}