{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:27:05Z","timestamp":1781587625082,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006374","name":"National University of Defense Technology","doi-asserted-by":"publisher","award":["2023ZD0121101"],"award-info":[{"award-number":["2023ZD0121101"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733429","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:31:39Z","timestamp":1750876299000},"page":"1569-1578","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Self-supervised Bidirectional Synchronization Estimation for Multimodal Deepfake Detection with Short-term Dependency"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1245-9487","authenticated-orcid":false,"given":"Man","family":"Xiao","sequence":"first","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Computing, College of Computer Science and Technology, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7055-3917","authenticated-orcid":false,"given":"Jianbin","family":"Ye","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Computing, College of Computer Science and Technology, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9953-8438","authenticated-orcid":false,"given":"Bo","family":"Liu","sequence":"additional","affiliation":[{"name":"Strategic Assessments and Consultation Institute, Academy of Military Science, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5151-3381","authenticated-orcid":false,"given":"Zijian","family":"Gao","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Computing, College of Computer Science and Technology, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Computing, College of Computer Science and Technology, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8949-5967","authenticated-orcid":false,"given":"Xiaodong","family":"Wang","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Computing, College of Computer Science and Technology, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Keshigeyan Chandrasegaran, Guimeng Liu, and Ngai-Man Cheung.","author":"Abdollahzadeh Milad","year":"2023","unstructured":"Milad Abdollahzadeh, Touba Malekzadeh, Christopher TH Teo, Keshigeyan Chandrasegaran, Guimeng Liu, and Ngai-Man Cheung. 2023. A survey on generative modeling with limited data, few shots, and zero shot. arXiv preprint arXiv:2307.14397 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"Senior Andrew, Vinyals Oriol, and Zisserman Andrew.","author":"Afouras Triantafyllos","year":"2018","unstructured":"Triantafyllos Afouras, Chung Joon Son, Senior Andrew, Vinyals Oriol, and Zisserman Andrew. 2018. Deep audio-visual speech recognition. In IEEE Transactions on Pattern Analysis and Machine Intelligence. 8717--8727."},{"key":"e_1_3_2_1_3_1","volume-title":"In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 5039--5049","author":"Alexandros Haliassos","year":"2021","unstructured":"Haliassos Alexandros, Vougioukas Konstantinos, Petridis Stavros, and Pantic Maja. 2021. Lips don't lie: A generalisable and robust approach to face forgery detection. In In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 5039--5049."},{"key":"e_1_3_2_1_4_1","volume-title":"In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 14950--14962","author":"Alexandros Haliassos","year":"2022","unstructured":"Haliassos Alexandros, Mira Rodrigo, Petridis Stavros, and Pantic Maja. 2022. Leveraging real talking faces via self-supervision for robust forgery detection. In In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 14950--14962."},{"key":"e_1_3_2_1_5_1","volume-title":"In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 1--11","author":"Andreas Rossler","year":"2019","unstructured":"Rossler Andreas, Cozzolino Davide, Verdoliva Luisa, Riess Christian, Thies Justus, and Nie\u00dfner Matthias. 2019. Faceforensics: Learning to detect manipulated facial images. In In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 1--11."},{"key":"e_1_3_2_1_6_1","volume-title":"Exposing the Deception: Uncovering More Forgery Clues for Deepfake Detection. CoRR","author":"Ba Zhongjie","year":"2024","unstructured":"Zhongjie Ba, Qingyu Liu, Zhenguang Liu, Shuang Wu, Feng Lin, Li Lu, and Kui Ren. 2024. Exposing the Deception: Uncovering More Forgery Clues for Deepfake Detection. CoRR, Vol. abs\/2403.01786 (2024). showeprint[arXiv]2403.01786"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00435"},{"key":"e_1_3_2_1_8_1","volume-title":"Audio-visual synchronisation in the wild. arXiv preprint arXiv:2112.04432","author":"Chen Honglie","year":"2021","unstructured":"Honglie Chen, Weidi Xie, Triantafyllos Afouras, Arsha Nagrani, Andrea Vedaldi, and Andrew Zisserman. 2021. Audio-visual synchronisation in the wild. arXiv preprint arXiv:2112.04432 (2021)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3625231"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2024.3409054"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00101"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2024.3369711"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01011"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i16.33842"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106513"},{"key":"e_1_3_2_1_16_1","volume-title":"Advances in Neural Information Processing Systems","author":"Gao Zijian","year":"2024","unstructured":"Zijian Gao, Xingxing Zhang, Kele Xu, Xinjun Mao, and Huaimin Wang. 2024b. Stabilizing Zero-Shot Prediction: A Novel Antidote to Forgetting in Continual Vision-Language Tasks. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang (Eds.), Vol. 37. Curran Associates, Inc., 128462--128488. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/file\/e7feb9dbd9a94b6c552fc403fcebf2ef-Paper-Conference.pdf"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652027"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01453"},{"key":"e_1_3_2_1_19_1","volume-title":"Guo Zhaohan, and Gheshlaghi Azar Mohammad.","author":"Jean-Bastien Grill","year":"2020","unstructured":"Grill Jean-Bastien, Strub Florian, Altch\u00b4e Florent, Tallec Corentin, Pierre Richemond, Buchatskaya Elena, Doersch Carl, Avila Pires Bernardo, Guo Zhaohan, and Gheshlaghi Azar Mohammad. 2020. Bootstrap your own latent-a new approach to self-supervised learning. In Advances in neural information processing systems. 21271--21284."},{"key":"e_1_3_2_1_20_1","volume-title":"Proc. Interspeech","author":"Son Chung Joon","year":"2018","unstructured":"Chung Joon Son, Nagrani Arsha, and Zisserman Andrew. 2018. Voxceleb2: Deep speaker recognition. In Proc. Interspeech 2018. 1086--1090."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Chatfield Ken Simonyan Karen Vedaldi Andrea and Zisserman Andrew. 2014. Return of the devil in the details: Delving deep into convolutional nets. In arXiv preprint arXiv:1405.3531. 4.","DOI":"10.5244\/C.28.6"},{"key":"e_1_3_2_1_22_1","volume-title":"In Proceedings of the IEEE conference on Computer Vision and Pattern Recognition. 6546--6555","author":"Kensho Hara","year":"2018","unstructured":"Hara Kensho, Kataoka Hirokatsu, and Satoh Yutaka. 2018. Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet?. In In Proceedings of the IEEE conference on Computer Vision and Pattern Recognition. 6546--6555."},{"key":"e_1_3_2_1_23_1","volume-title":"Simon","author":"Shahroz Tariq","year":"2021","unstructured":"Khalid, Hasam, Tariq Shahroz, Kim Minha, and S. Woo. Simon. 2021. FakeAVCeleb: A novel audio-video multimodal deepfake dataset. In Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks. 1--8."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01057"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02345"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02559"},{"key":"e_1_3_2_1_27_1","volume-title":"In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 12873--12883","author":"Patrick Esser","year":"2021","unstructured":"Esser Patrick, Rombach Robin, and Ommer Bjorn. 2021. Taming transformers for high-resolution image synthesis. In In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 12873--12883."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00070"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3356814"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3356814"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 993--1000","author":"Raza Muhammad Anas","year":"2023","unstructured":"Muhammad Anas Raza and Khalid Mahmood Malik. 2023. Multimodaltrace: Deepfake detection using audiovisual representation learning. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 993--1000."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","unstructured":"Chang-Sung Sung Jun-Cheng Chen and Chu-Song Chen. 2023. Hearing and Seeing Abnormality: Self-Supervised Audio-Visual Mutual Learning for Deepfake Detection. In ICASSP 2023 - 2023 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP). 1--5. doi:10.1109\/ICASSP49357.2023.10095247","DOI":"10.1109\/ICASSP49357.2023.10095247"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680895"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01408"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02116-5"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3269841"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02054-2"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3672566"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02128-1"},{"key":"e_1_3_2_1_40_1","volume-title":"In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 15044--15054","author":"Yinglin Zheng","year":"2021","unstructured":"Zheng Yinglin, Bao Jianmin, Chen Dong, Zeng Ming, and Wen Fang. 2021. Exploring temporal coherence for more general video face forgery detection. In In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 15044--15054."},{"key":"e_1_3_2_1_41_1","volume-title":"In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 14800--14809","author":"Yipin Zhou","year":"2021","unstructured":"Zhou Yipin and Lim Ser-Nam. 2021. Joint audio-visual deepfake detection. In In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 14800--14809."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2024.3364679"},{"key":"e_1_3_2_1_43_1","volume-title":"Pvass-mdd: predictive visual-audio alignment self-supervision for multimodal deepfake detection","author":"Yu Yang","year":"2023","unstructured":"Yang Yu, Xiaolong Liu, Rongrong Ni, Siyuan Yang, Yao Zhao, and Alex C Kot. 2023. Pvass-mdd: predictive visual-audio alignment self-supervision for multimodal deepfake detection. IEEE Transactions on Circuits and Systems for Video Technology (2023)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3625100"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3625100"},{"key":"e_1_3_2_1_46_1","volume-title":"Audio-Visual Contrastive Pre-train for Face Forgery Detection. ACM Transactions on Multimedia Computing, Communications and Applications","author":"Zhao Hanqing","year":"2024","unstructured":"Hanqing Zhao, Wenbo Zhou, Dongdong Chen, Weiming Zhang, Ying Guo, Zhen Cheng, Pengfei Yan, and Nenghai Yu. 2024. Audio-Visual Contrastive Pre-train for Face Forgery Detection. ACM Transactions on Multimedia Computing, Communications and Applications (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"Eng Siong Chng, and Deepu Rajan","author":"Zou Heqing","year":"2024","unstructured":"Heqing Zou, Meng Shen, Yuchen Hu, Chen Chen, Eng Siong Chng, and Deepu Rajan. 2024. Cross-Modality and Within-Modality Regularization for Audio-Visual Deepfake Detection. In ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4900--4904."}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733429","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:12:54Z","timestamp":1755749574000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733429"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":47,"alternative-id":["10.1145\/3731715.3733429","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733429","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}