{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,24]],"date-time":"2025-08-24T01:39:53Z","timestamp":1755999593034,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Slovenian Research and Innovation Agency ARIS","award":["J2-50065, P0-0250 and P2-0214"],"award-info":[{"award-number":["J2-50065, P0-0250 and P2-0214"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3689092.3689410","type":"proceedings-article","created":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T18:33:17Z","timestamp":1729708397000},"page":"24-29","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["W-TDL: Window-Based Temporal Deepfake Localization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0149-3949","authenticated-orcid":false,"given":"Luka","family":"Dragar","sequence":"first","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4491-2744","authenticated-orcid":false,"given":"Peter","family":"Rot","sequence":"additional","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9744-4035","authenticated-orcid":false,"given":"Peter","family":"Peer","sequence":"additional","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3385-5780","authenticated-orcid":false,"given":"Vitomir","family":"\u0160truc","sequence":"additional","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8669-2671","authenticated-orcid":false,"given":"Borut","family":"Batagelj","sequence":"additional","affiliation":[{"name":"University of Ljubljana, Ljubljana, Slovenia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. CoRR","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Henry Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. CoRR, Vol. abs\/2006.11477 (2020). showeprint[arXiv]2006.11477 https:\/\/arxiv.org\/abs\/2006.11477"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 27th Computer Vision Winter Workshop (CVWW 2024","author":"Marko","year":"2024","unstructured":"Marko Brodari?, Vitomir Struc, and Peter Peer. 2024. Cross-dataset deepfake detection: evaluating the generalization capabilities of modern deepfake detectors. In Proceedings of the 27th Computer Vision Winter Workshop (CVWW 2024). Slovensko dru?tvo za razpoznavanje vzorcev = Slovenian Pattern Recognition Society, 47--56. https:\/\/cvww2024.sdrv.si\/proceedings\/"},{"key":"e_1_3_2_1_3_1","volume-title":"Munawar Hayat, Abhinav Dhall, and Kalin Stefanov.","author":"Cai Zhixi","year":"2023","unstructured":"Zhixi Cai, Shreya Ghosh, Aman Pankaj Adatia, Munawar Hayat, Abhinav Dhall, and Kalin Stefanov. 2023. AV-Deepfake1M: A Large-Scale LLM-Driven Audio-Visual Deepfake Dataset. arXiv preprint arXiv:2311.15308 (2023)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103818"},{"key":"e_1_3_2_1_5_1","volume-title":"Eren G\u00f6lge, and Moacir Antonelli Ponti.","author":"Casanova Edresson","year":"2021","unstructured":"Edresson Casanova, Julian Weber, Christopher Shulby, Arnaldo C\u00e2ndido J\u00fanior, Eren G\u00f6lge, and Moacir Antonelli Ponti. 2021. YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for everyone. CoRR, Vol. abs\/2112.02418 (2021). showeprint[arXiv]2112.02418 https:\/\/arxiv.org\/abs\/2112.02418"},{"key":"e_1_3_2_1_6_1","volume-title":"VoxCeleb2: Deep Speaker Recognition. CoRR","author":"Chung Joon Son","year":"2018","unstructured":"Joon Son Chung, Arsha Nagrani, and Andrew Zisserman. 2018. VoxCeleb2: Deep Speaker Recognition. CoRR, Vol. abs\/1806.05622 (2018). showeprint[arXiv]1806.05622 http:\/\/arxiv.org\/abs\/1806.05622"},{"key":"e_1_3_2_1_7_1","volume-title":"The deepfake detection challenge (dfdc) dataset. arXiv preprint arXiv:2006.07397","author":"Dolhansky Brian","year":"2020","unstructured":"Brian Dolhansky, Joanna Bitton, Ben Pflaum, Jikuo Lu, Russ Howes, Menglin Wang, and Cristian Canton Ferrer. 2020. The deepfake detection challenge (dfdc) dataset. arXiv preprint arXiv:2006.07397 (2020)."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 32nd International Electrotechnical and Computer Science Conference ERK 2023. Slovenska sekcija IEEE; Fakulteta za elektrotehniko, Portoro?, Slovenija, 363--366","author":"Dragar Luka","year":"2023","unstructured":"Luka Dragar, Peter Peer, Vitomir Struc, and Borut Batagelj. 2023. Beyond detection: visual realism assessment of deepfakes. In Proceedings of the 32nd International Electrotechnical and Computer Science Conference ERK 2023. Slovenska sekcija IEEE; Fakulteta za elektrotehniko, Portoro?, Slovenija, 363--366. https:\/\/erk.fe.uni-lj.si\/2023\/papers\/dragar(beyond_detection_).pdf"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/AVSS.2018.8639163"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCB54206.2022.10007950"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00044"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW60836.2024.00115"},{"key":"e_1_3_2_1_14_1","volume-title":"Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech. CoRR","author":"Kim Jaehyeon","year":"2021","unstructured":"Jaehyeon Kim, Jungil Kong, and Juhee Son. 2021. Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech. CoRR, Vol. abs\/2106.06103 (2021). showeprint[arXiv]2106.06103 https:\/\/arxiv.org\/abs\/2106.06103"},{"key":"e_1_3_2_1_15_1","volume-title":"The creation and detection of deepfakes: A survey. ACM computing surveys (CSUR)","author":"Mirsky Yisroel","year":"2021","unstructured":"Yisroel Mirsky and Wenke Lee. 2021. The creation and detection of deepfakes: A survey. ACM computing surveys (CSUR), Vol. 54, 1 (2021), 1--41."},{"key":"e_1_3_2_1_16_1","volume-title":"DFGC-VRA: DeepFake Game Competition on Visual Realism Assessment. In 2023 IEEE International Joint Conference on Biometrics (IJCB). IEEE, 1--9.","author":"Peng Bo","year":"2023","unstructured":"Bo Peng, Xianyun Sun, Caiyong Wang, Wei Wang, Jing Dong, Zhenan Sun, Rongyu Zhang, Heng Cong, Lingzhi Fu, Hao Wang, et al. 2023. DFGC-VRA: DeepFake Game Competition on Visual Realism Assessment. In 2023 IEEE International Joint Conference on Biometrics (IJCB). IEEE, 1--9."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413707"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.5555\/3618408.3619590"},{"key":"e_1_3_2_1_19_1","volume-title":"Beddhu Murali, and Andrew H Sung.","author":"Rana Md Shohel","year":"2022","unstructured":"Md Shohel Rana, Mohammad Nur Nobi, Beddhu Murali, and Andrew H Sung. 2022. Deepfake detection: A systematic literature review. IEEE access, Vol. 10 (2022), 25494--25513."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01408"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101114"},{"key":"e_1_3_2_1_22_1","unstructured":"Yuankun Xie Haonan Cheng Yutian Wang and Long Ye. 2023. An Efficient Temporary Deepfake Location Approach Based Embeddings for Partially Spoofed Audio Detection. arxiv: 2309.03036 [cs.SD] https:\/\/arxiv.org\/abs\/2309.03036"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1049\/bme2.12031"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3613767"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11733-y"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 2nd International Workshop on Multimodal and Responsible Affective Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689092.3689410","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3689092.3689410","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:59:17Z","timestamp":1755914357000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3689092.3689410"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":25,"alternative-id":["10.1145\/3689092.3689410","10.1145\/3689092"],"URL":"https:\/\/doi.org\/10.1145\/3689092.3689410","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}