{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T08:53:36Z","timestamp":1781600016687,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T00:00:00Z","timestamp":1781568000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"European Union","award":["D43C22003080001"],"award-info":[{"award-number":["D43C22003080001"]}]},{"name":"European Union","award":["D43C22003050001"],"award-info":[{"award-number":["D43C22003050001"]}]},{"name":"Italian Ministry of Education, University, and Research","award":["FOSTERER project"],"award-info":[{"award-number":["FOSTERER project"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,17]]},"DOI":"10.1145\/3785353.3815078","type":"proceedings-article","created":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T08:11:30Z","timestamp":1781597490000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Forensic Similarity for Speech Deepfakes"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-2483-4366","authenticated-orcid":false,"given":"Viola","family":"Negroni","sequence":"first","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5163-3364","authenticated-orcid":false,"given":"Davide","family":"Salvi","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3217-9952","authenticated-orcid":false,"given":"Daniele Ugo","family":"Leonzio","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0406-0222","authenticated-orcid":false,"given":"Paolo","family":"Bestagini","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1990-9869","authenticated-orcid":false,"given":"Stefano","family":"Tubaro","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,16]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Henry Ajder Giorgio Patrini Francesco Cavalli and Laurence Cullen. 2019. The state of deepfakes: Landscape threats and impact. Amsterdam: Deeptrace 27 (2019)."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"crossref","unstructured":"Irene Amerini Mauro Barni Sebastiano Battiato Paolo Bestagini Giulia Boato Vittoria Bruni Roberto Caldelli Francesco De\u00a0Natale Rocco De\u00a0Nicola Luca Guarnera et\u00a0al. 2025. Deepfake media forensics: Status and future challenges. Journal of Imaging 11 3 (2025) 73.","DOI":"10.3390\/jimaging11030073"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"crossref","unstructured":"Arun Babu Changhan Wang Andros Tjandra Kushal Lakhotia Qiantong Xu Naman Goyal Kritika Singh Patrick von Platen Yatharth Saraf Juan Pino Alexei Baevski Alexis Conneau and Michael Auli. 2021. XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale. arXiv abs\/2111.09296 (2021).","DOI":"10.21437\/Interspeech.2022-143"},{"key":"e_1_3_3_1_5_2","unstructured":"Alexei Baevski Yuhao Zhou Abdelrahman Mohamed and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 (2020) 12449\u201312460."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Huda Barakat Oytun Turk and Cenk Demiroglu. 2024. Deep learning-based expressive speech synthesis: a systematic review of approaches challenges and resources. EURASIP Journal on Audio Speech and Music Processing 2024 1 (2024) 11.","DOI":"10.1186\/s13636-024-00329-7"},{"key":"e_1_3_3_1_7_2","volume-title":"Proc. ACM Workshop on Information Hiding and Multimedia Security","author":"Bhagtani Kratika","year":"2023","unstructured":"Kratika Bhagtani, Emily\u00a0R Bartusiak, Amit Kumar\u00a0Singh Yadav, Paolo Bestagini, and Edward\u00a0J Delp. 2023. Synthesized speech attribution using the patchout spectrogram attribution transformer. In Proc. ACM Workshop on Information Hiding and Multimedia Security."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Clara Borrelli Paolo Bestagini Fabio Antonacci Augusto Sarti and Stefano Tubaro. 2021. Synthetic speech detection through short-term and long-term prediction traces. EURASIP Journal on Information Security 2021 (2021).","DOI":"10.1186\/s13635-021-00116-3"},{"key":"e_1_3_3_1_9_2","first-page":"1","volume-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Cai Zexin","year":"2023","unstructured":"Zexin Cai, Weiqing Wang, and Ming Li. 2023. Waveform boundary detection for partially spoofed audio. In ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1\u20135."},{"key":"e_1_3_3_1_10_2","volume-title":"IEEE Computer society conference on Computer Vision and Pattern Recognition (CVPR)","author":"Chopra Sumit","year":"2005","unstructured":"Sumit Chopra, Raia Hadsell, and Yann LeCun. 2005. Learning a similarity metric discriminatively, with application to face verification. In IEEE Computer society conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Davide Cozzolino and Luisa Verdoliva. 2019. Noiseprint: A CNN-based camera model fingerprint. IEEE Transactions on Information Forensics and Security 15 (2019) 144\u2013159.","DOI":"10.1109\/TIFS.2019.2916364"},{"key":"e_1_3_3_1_12_2","volume-title":"Proc. Interspeech 2025","author":"Falez Pierre","year":"2025","unstructured":"Pierre Falez, Tony Marteau, Damien Lolive, and Arnaud Delhay. 2025. Audio Deepfake Source Tracing using Multi-Attribute Open-Set Identification and Verification. In Proc. Interspeech 2025."},{"key":"e_1_3_3_1_13_2","volume-title":"IEEE conference on Computer Vision and Pattern Recognition (CVPR)","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep residual learning for image recognition. In IEEE conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_3_1_14_2","unstructured":"Keith Ito and Linda Johnson. 2017. The LJ Speech Dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/."},{"key":"e_1_3_3_1_15_2","first-page":"6367","volume-title":"ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP)","author":"Jung Jee-weon","year":"2022","unstructured":"Jee-weon Jung, Hee-Soo Heo, Hemlata Tak, Hye-jin Shim, Joon\u00a0Son Chung, Bong-Jin Lee, Ha-Jin Yu, and Nicholas Evans. 2022. Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks. In ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, 6367\u20136371."},{"key":"e_1_3_3_1_16_2","volume-title":"Proceedings of Interspeech 2022","author":"Kawa Piotr","year":"2022","unstructured":"Piotr Kawa, Marcin Plata, and Piotr Syga. 2022. Attack Agnostic Dataset: Towards Generalization and Stabilization of Audio DeepFake Detection. In Proceedings of Interspeech 2022."},{"key":"e_1_3_3_1_17_2","volume-title":"Proc. INTERSPEECH","author":"Klein Nicholas","year":"2024","unstructured":"Nicholas Klein, Tianxiang Chen, Hemlata Tak, Ricardo Casal, and Elie Khoury. 2024. Source Tracing of Audio Deepfake Systems. In Proc. INTERSPEECH."},{"key":"e_1_3_3_1_18_2","volume-title":"Proc. Interspeech 2025","author":"Klein Nicholas","year":"2025","unstructured":"Nicholas Klein, Hemlata Tak, and Elie Khoury. 2025. Open-Set Source Tracing of Audio Deepfake Systems. In Proc. Interspeech 2025."},{"key":"e_1_3_3_1_19_2","volume-title":"Proc. Interspeech 2025","author":"Koutsianos Dimitrios","year":"2025","unstructured":"Dimitrios Koutsianos, Stavros Zacharopoulos, Yannis Panagakis, and Themos Stafylakis. 2025. Synthetic Speech Source Tracing using Metric Learning. In Proc. Interspeech 2025."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Marie-Helen Maras and Alex Alexandrou. 2019. Determining authenticity of video evidence in the age of artificial intelligence and in the wake of Deepfake videos. The International Journal of Evidence & Proof 23 3 (2019) 255\u2013262.","DOI":"10.1177\/1365712718807226"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Owen Mayer and Matthew\u00a0C Stamm. 2019. Forensic similarity for digital images. IEEE Transactions on Information Forensics and Security 15 (2019) 1331\u20131346.","DOI":"10.1109\/TIFS.2019.2924552"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Jagabandhu Mishra Manasi Chhibber Hye-jin Shim and Tomi\u00a0H Kinnunen. 2026. Towards explainable spoofed speech attribution and detection: a probabilistic approach for characterizing speech synthesizer components. Computer Speech & Language 95 (2026).","DOI":"10.1016\/j.csl.2025.101840"},{"key":"e_1_3_3_1_23_2","unstructured":"Nicolas M\u00fcller. 2024. Using MLAAD for Source Tracing of Audio Deepfakes. https:\/\/deepfake-total.com\/sourcetracing."},{"key":"e_1_3_3_1_24_2","volume-title":"Proceedings of Interspeech 2024","author":"M\u00fcller Nicolas\u00a0M.","year":"2024","unstructured":"Nicolas\u00a0M. M\u00fcller, Nicholas Evans, Hemlata Tak, Philip Sperl, and Konstantin B\u00f6ttinger. 2024. Harder or Different? Understanding Generalization of Audio Deepfake Detection. In Proceedings of Interspeech 2024."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Nicolas\u00a0M M\u00fcller Piotr Kawa Wei\u00a0Herng Choong Edresson Casanova Eren G\u00f6lge Thorsten M\u00fcller Piotr Syga Philip Sperl and Konstantin B\u00f6ttinger. 2024. MLAAD: The Multi-Language Audio Anti-Spoofing Dataset. IEEE International Joint Conference on Neural Networks (IJCNN) (2024).","DOI":"10.1109\/IJCNN60899.2024.10650962"},{"key":"e_1_3_3_1_26_2","volume-title":"Proc. Interspeech 2025","author":"Negroni Viola","year":"2025","unstructured":"Viola Negroni, Davide Salvi, Paolo Bestagini, and Stefano Tubaro. 2025. Source verification for speech deepfakes. In Proc. Interspeech 2025."},{"key":"e_1_3_3_1_27_2","volume-title":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Panayotov Vassil","year":"2015","unstructured":"Vassil Panayotov, Guoguo Chen, Daniel Povey, and Sanjeev Khudanpur. 2015. Librispeech: an asr corpus based on public domain audio books. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","first-page":"1021","DOI":"10.1109\/SLT.2018.8639585","volume-title":"2018 IEEE spoken language technology workshop (SLT)","author":"Ravanelli Mirco","year":"2018","unstructured":"Mirco Ravanelli and Yoshua Bengio. 2018. Speaker recognition from raw waveform with sincnet. In 2018 IEEE spoken language technology workshop (SLT). IEEE, 1021\u20131028."},{"key":"e_1_3_3_1_29_2","volume-title":"IEEE International Workshop on Information Forensics and Security (WIFS)","author":"Salvi Davide","year":"2022","unstructured":"Davide Salvi, Paolo Bestagini, and Stefano Tubaro. 2022. Exploring the synthetic speech attribution problem through data-driven detectors. In IEEE International Workshop on Information Forensics and Security (WIFS)."},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Davide Salvi Brian Hosler Paolo Bestagini Matthew\u00a0C Stamm and Stefano Tubaro. 2023. TIMIT-TTS: a Text-to-Speech Dataset for Multimodal Synthetic Media Detection. IEEE Access (2023).","DOI":"10.1109\/ACCESS.2023.3276480"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","first-page":"329","DOI":"10.1109\/IEEECONF60004.2024.10942913","volume-title":"2024 58th Asilomar Conference on Signals, Systems, and Computers","author":"Salvi Davide","year":"2024","unstructured":"Davide Salvi, Amit Kumar\u00a0Singh Yadav, Kratika Bhagtani, Viola Negronil, Paolo Bestagini, and Edward\u00a0J Delp. 2024. Comparative Analysis of ASR Methods for Speech Deepfake Detection. In 2024 58th Asilomar Conference on Signals, Systems, and Computers. IEEE, 329\u2013333."},{"key":"e_1_3_3_1_32_2","volume-title":"The VidTIMIT database","author":"Sanderson Conrad","year":"2002","unstructured":"Conrad Sanderson. 2002. The VidTIMIT database. Technical Report. IDIAP."},{"key":"e_1_3_3_1_33_2","volume-title":"Proc. Interspeech 2025","author":"Stan Adriana","year":"2025","unstructured":"Adriana Stan, David Combei, Dan Oneata, and Horia Cucu. 2025. TADA: Training-free attribution and out-of-domain detection of audio deepfakes. In Proc. Interspeech 2025."},{"key":"e_1_3_3_1_34_2","first-page":"6369","volume-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Tak Hemlata","year":"2021","unstructured":"Hemlata Tak, Jose Patino, Massimiliano Todisco, Andreas Nautsch, Nicholas Evans, and Anthony Larcher. 2021. End-to-end anti-spoofing with rawnet2. In ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6369\u20136373."},{"key":"e_1_3_3_1_35_2","volume-title":"Proc. INTERSPEECH","author":"Todisco Massimiliano","year":"2019","unstructured":"Massimiliano Todisco, Xin Wang, Ville Vestman, Md Sahidullah, H\u00e9ctor Delgado, Andreas Nautsch, Junichi Yamagishi, Nicholas Evans, Tomi Kinnunen, and Kong\u00a0Aik Lee. 2019. ASVspoof 2019: Future horizons in spoofed and fake audio detection. In Proc. INTERSPEECH."},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"crossref","unstructured":"Cristian Vaccari and Andrew Chadwick. 2020. Deepfakes and disinformation: Exploring the impact of synthetic political video on deception uncertainty and trust in news. Social media+ society 6 1 (2020) 2056305120903408.","DOI":"10.1177\/2056305120903408"},{"key":"e_1_3_3_1_37_2","unstructured":"Christophe Veaux Junichi Yamagishi Kirsten MacDonald et\u00a0al. 2016. Superseded-CSTR VCTK Corpus: English Multi-Speaker Corpus for CSTR Voice Cloning Toolkit. University of Edinburgh. The Centre for Speech Technology Research (CSTR) (2016)."},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"crossref","first-page":"115","DOI":"10.1109\/ISCSLP57327.2022.10037999","volume-title":"2022 13th International Symposium on Chinese Spoken Language Processing (ISCSLP)","author":"Wang Lei","year":"2022","unstructured":"Lei Wang, Benedict Yeoh, and Jun\u00a0Wah Ng. 2022. Synthetic voice detection and audio splicing detection using se-res2net-conformer architecture. In 2022 13th International Symposium on Chinese Spoken Language Processing (ISCSLP). IEEE, 115\u2013119."},{"key":"e_1_3_3_1_39_2","first-page":"9236","volume-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Wu Haibin","year":"2022","unstructured":"Haibin Wu, Heng-Cheng Kuo, Naijun Zheng, Kuo-Hsuan Hung, Hung-Yi Lee, Yu Tsao, Hsin-Min Wang, and Helen Meng. 2022. Partially fake audio detection by self-attention-based fake span discovery. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 9236\u20139240."},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"crossref","unstructured":"Xiang Wu Ran He Zhenan Sun and Tieniu Tan. 2018. A light CNN for deep face representation with noisy labels. IEEE Transactions on Information Forensics and Security 13 11 (2018).","DOI":"10.1109\/TIFS.2018.2833032"},{"key":"e_1_3_3_1_41_2","first-page":"11171","volume-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Yadav Amit Kumar\u00a0Singh","year":"2024","unstructured":"Amit Kumar\u00a0Singh Yadav, Kratika Bhagtani, Sriram Baireddy, Paolo Bestagini, Stefano Tubaro, and Edward\u00a0J Delp. 2024. Mdrt: Multi-Domain Synthetic Speech Localization. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 11171\u201311175."},{"key":"e_1_3_3_1_42_2","first-page":"9216","volume-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Yi Jiangyan","year":"2022","unstructured":"Jiangyan Yi, Ruibo Fu, Jianhua Tao, Shuai Nie, Haoxin Ma, Chenglong Wang, Tao Wang, Zhengkun Tian, Ye Bai, Cunhang Fan, et\u00a0al. 2022. Add 2022: the first audio deep synthesis detection challenge. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 9216\u20139220."},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"crossref","unstructured":"Lin Zhang Xin Wang Erica Cooper Nicholas Evans and Junichi Yamagishi. 2022. The partialspoof database and countermeasures for the detection of short fake speech segments embedded in an utterance. IEEE\/ACM Transactions on Audio Speech and Language Processing 31 (2022) 813\u2013825.","DOI":"10.1109\/TASLP.2022.3233236"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"crossref","unstructured":"Lin Zhang Xin Wang Erica Cooper and Junichi Yamagishi. 2021. Multi-task learning in utterance-level and segmental-level spoof detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.14132 (2021).","DOI":"10.21437\/ASVSPOOF.2021-2"},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"crossref","unstructured":"Lin Zhang Xin Wang Erica Cooper Junichi Yamagishi Jose Patino and Nicholas Evans. 2021. An initial investigation for detecting partially spoofed audio. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.02518 (2021).","DOI":"10.21437\/Interspeech.2021-738"}],"event":{"name":"IH&MMSec '26: ACM Workshop on Information Hiding and Multimedia Security","location":"Firenze Italy","acronym":"IH&MMSec '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 ACM Workshop on Information Hiding and Multimedia Security"],"original-title":[],"deposited":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T08:12:09Z","timestamp":1781597529000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3785353.3815078"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,16]]},"references-count":44,"alternative-id":["10.1145\/3785353.3815078","10.1145\/3785353"],"URL":"https:\/\/doi.org\/10.1145\/3785353.3815078","relation":{},"subject":[],"published":{"date-parts":[[2026,6,16]]},"assertion":[{"value":"2026-06-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}