{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T19:01:33Z","timestamp":1782846093630,"version":"3.54.5"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T00:00:00Z","timestamp":1761696000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T00:00:00Z","timestamp":1761696000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"SERICS under the MUR National Recovery and Resilience Plan funded by the European Union - NextGenerationEU.","award":["PE00000014"],"award-info":[{"award-number":["PE00000014"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["EURASIP J. on Info. Security"],"DOI":"10.1186\/s13635-025-00217-3","type":"journal-article","created":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T13:28:06Z","timestamp":1761744486000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["ASTDT: an Interpretable Adaptive Spectro-Temporal Diffusion Transformer for audio deepfake detection"],"prefix":"10.1186","volume":"2025","author":[{"given":"Taiba\u00a0Maijd","family":"Wani","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Syed\u00a0Asif\u00a0Ahmad","family":"Qadri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Arselan","family":"Ashraf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Irene","family":"Amerini","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,29]]},"reference":[{"key":"217_CR1","unstructured":"DataReportal. Digital 2024: Global Overview Report. 2024. https:\/\/datareportal.com\/reports\/digital-2024-global-overviewreport Online. Accessed 2024"},{"key":"217_CR2","first-page":"279","volume":"15","author":"G Abiri","year":"2024","unstructured":"G. Abiri, Generative ai as digital media. Harv. J. Sports Ent. L. 15, 279 (2024)","journal-title":"Harv. J. Sports Ent. L."},{"issue":"1","key":"217_CR3","doi-asserted-by":"publisher","first-page":"549","DOI":"10.1007\/s42001-024-00250-1","volume":"7","author":"E Ferrara","year":"2024","unstructured":"E. Ferrara, Genai against humanity: nefarious applications of generative artificial intelligence and large language models. J. Comput. Soc. Sci. 7(1), 549\u2013569 (2024)","journal-title":"J. Comput. Soc. Sci."},{"key":"217_CR4","doi-asserted-by":"publisher","unstructured":"Garon Jon M. A practical introduction to generative ai, synthetic media, and the messages found in the latest medium. Synthetic Media, and the Messages Found in the Latest Medium.\u00a0SSRN. Electron. J. (2023). https:\/\/doi.org\/10.2139\/ssrn.4388437","DOI":"10.2139\/ssrn.4388437"},{"key":"217_CR5","doi-asserted-by":"publisher","unstructured":"M. Westerlund, The emergence of deepfake technology: a review. Technol. Innov. Manag. Rev. 9(11),\u00a040\u201353 (2019). https:\/\/doi.org\/10.22215\/timreview\/1282","DOI":"10.22215\/timreview\/1282"},{"key":"217_CR6","doi-asserted-by":"publisher","unstructured":"H. Pandian, N. Rawindaran,\u00a0Redefining Reality in Political Propaganda: Exploring the Impact of Superimposed Deepfakes in Misinformation. Data Protection: The Wake of AI and Machine Learning (Springer,\u00a0Cham, 2024), pp. 47\u201361.\u00a0https:\/\/doi.org\/10.1007\/978-3-031-76473-8_3","DOI":"10.1007\/978-3-031-76473-8_3"},{"key":"217_CR7","doi-asserted-by":"crossref","unstructured":"M.E. Myers, in Understanding Media Psychology (Routledge, 2021), pp. 161\u2013181","DOI":"10.4324\/9781003055648-8"},{"issue":"22","key":"217_CR8","first-page":"3242","volume":"97","author":"M Albahar","year":"2019","unstructured":"M. Albahar, J. Almalki, Deepfakes: threats and countermeasures systematic review. J. Theor. Appl. Inf. Technol. 97(22), 3242\u20133250 (2019)","journal-title":"J. Theor. Appl. Inf. Technol."},{"issue":"5","key":"217_CR9","doi-asserted-by":"publisher","first-page":"910","DOI":"10.1109\/JSTSP.2020.3002101","volume":"14","author":"L Verdoliva","year":"2020","unstructured":"L. Verdoliva, Media forensics and deepfakes: an overview. IEEE J. Sel. Top. Signal Process. 14(5), 910\u2013932 (2020)","journal-title":"IEEE J. Sel. Top. Signal Process."},{"issue":"4","key":"217_CR10","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1561\/0600000096","volume":"12","author":"I Amerini","year":"2021","unstructured":"I. Amerini, A. Anagnostopoulos, L. Maiano, L.R. Celsi, Deep learning for multimedia forensics. Foundations and Trends\u00ae in Computer Graphics and Vision 12(4), 309\u2013457 (2021)","journal-title":"Foundations and Trends\u00ae in Computer Graphics and Vision"},{"issue":"4","key":"217_CR11","doi-asserted-by":"publisher","first-page":"3974","DOI":"10.1007\/s10489-022-03766-z","volume":"53","author":"M Masood","year":"2023","unstructured":"M. Masood, M. Nawaz, K.M. Malik, A. Javed, A. Irtaza, H. Malik, Deepfakes generation and detection: state-of-the-art, open challenges, countermeasures, and way forward. Appl. Intell. 53(4), 3974\u20134026 (2023)","journal-title":"Appl. Intell."},{"key":"217_CR12","doi-asserted-by":"publisher","unstructured":"J.T. Hancock, J.N. Bailenson, The social impact of deepfakes. Cyberpsychol. Behav Soc Netw. 24(3), 149\u2013152 (2021). https:\/\/doi.org\/10.1089\/cyber.2021.29208.jth","DOI":"10.1089\/cyber.2021.29208.jth"},{"key":"217_CR13","unstructured":"Cyber Risk Leaders. Audio Deepfakes Flood Social Media Platforms. 2024. https:\/\/cyberriskleaders.com\/audio-deepfakes-flood-social-media-platforms\/ Online. Accessed 2024"},{"key":"217_CR14","doi-asserted-by":"crossref","unstructured":"M. Mylrea,\u00a0The Generative AI Weapon of Mass Destruction: Evolving Disinformation Threats, Vulnerabilities, and Mitigation. in Interdependent Human-Machine Teams (Academic Press, Elsevier,\u00a0Amsterdam, 2025), pp. 315\u2013347","DOI":"10.1016\/B978-0-443-29246-0.00007-9"},{"key":"217_CR15","doi-asserted-by":"publisher","unstructured":"J. Davis,\u00a0Disinformation in the Era of Generative AI: Challenges, Detection Strategies, and Countermeasures. in Public Relations and the Rise of AI (Routledge,\u00a0New York, 2025), pp. 242\u2013269.\u00a0https:\/\/doi.org\/10.4324\/9781032671482","DOI":"10.4324\/9781032671482"},{"key":"217_CR16","unstructured":"L.F. Cervini, M. Emilia, M.V. Carro, An Overview of the Impact of GenAI and Deepfakes on Global Electoral Processes. (2024). https:\/\/www.ispionline.it\/en\/publication\/an-overview-of-the-impact-of-genai-and-deepfakes-on-global-electoral-processes-167584. Accessed 25 Oct 2024"},{"issue":"3\u20134","key":"217_CR17","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1561\/3300000048","volume":"6","author":"TM Wani","year":"2024","unstructured":"T.M. Wani, S.A.A. Qadri, F.A. Wani, I. Amerini, Navigating the soundscape of deception: a comprehensive survey on audio deepfake generation, detection, and future horizons. Foundations and Trends\u00ae in Privacy and Security 6(3\u20134), 153\u2013345 (2024)","journal-title":"Foundations and Trends\u00ae in Privacy and Security"},{"key":"217_CR18","doi-asserted-by":"publisher","DOI":"10.3389\/fdata.2022.1001063","volume":"5","author":"Z Khanjani","year":"2023","unstructured":"Z. Khanjani, G. Watson, V.P. Janeja, Audio deepfakes: a survey. Front. Big Data 5, 1001063 (2023)","journal-title":"Front. Big Data"},{"key":"217_CR19","doi-asserted-by":"publisher","first-page":"144497","DOI":"10.1109\/ACCESS.2023.3344653","volume":"11","author":"R Mubarak","year":"2023","unstructured":"R. Mubarak, T. Alsboui, O. Alshaikh, I. Inuwa-Dutse, S. Khan, S. Parkinson, A survey on the detection and impacts of deepfakes in visual, audio, and textual formats. IEEE Access 11, 144497\u2013144529 (2023)","journal-title":"IEEE Access"},{"key":"217_CR20","unstructured":"C. Wang, S. Chen, Y. Wu, Z. Zhang, L. Zhou, S. Liu, Z. Chen, Y. Liu, H. Wang, J. Li et al.,\u00a0Neural Codec Language Models Are Zero-Shot Text-to-Speech Synthesizers. (2023).\u00a0arXiv\u00a0preprint\u00a0arXiv:2301.02111.\u00a0https:\/\/arxiv.org\/abs\/2301.02111"},{"key":"217_CR21","unstructured":"A.v.d. Oord, S. Dieleman, H. Zen, K. Simonyan, O. Vinyals, A. Graves, N. Kalchbrenner, A. Senior, K. Kavukcuoglu, Wavenet: A generative model for raw audio.\u00a0Proceedings of the 9th ISCA Speech Synthesis Workshop (SSW9). p. 125, (2016).\u00a0arXiv\u00a0preprint\u00a0arXiv:1609.03499"},{"key":"217_CR22","unstructured":"E. Casanova, J. Weber, C.D. Shulby, A.C. Junior, E. G\u00f6lge, M.A. Ponti,\u00a0YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for Everyone.\u00a0Proceedings of the 14th International Conference on Machine Learning and Computing (ICMLC 2022)\u00a0(PMLR,\u00a0Cambridge, 2022), pp. 2709\u20132720.\u00a0https:\/\/proceedings.mlr.press\/vxxxx\/xx.html"},{"key":"217_CR23","doi-asserted-by":"publisher","unstructured":"J. Li, W. Tu, L. Xiao, Freevc: Towards high-quality text-free one-shot voice conversion. in ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (IEEE,\u00a0Rhodes Island, Greece, 2023), pp. 1\u20135.\u00a0https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10095191","DOI":"10.1109\/ICASSP49357.2023.10095191"},{"key":"217_CR24","first-page":"1753","volume":"107","author":"B Chesney","year":"2019","unstructured":"B. Chesney, D. Citron, Deep fakes: a looming challenge for privacy, democracy, and national security. Calif. L. Rev. 107, 1753 (2019)","journal-title":"Calif. L. Rev."},{"key":"217_CR25","doi-asserted-by":"publisher","unstructured":"S. Lyu, Deepfake the menace: mitigating the negative impacts of ai-generated content. Org. Cybersecur. J: Pract, Process People. 4(1),\u00a01\u201318 (2024)\u00a0https:\/\/doi.org\/10.1108\/OCJ-08-2022-0014","DOI":"10.1108\/OCJ-08-2022-0014"},{"issue":"7","key":"217_CR26","doi-asserted-by":"publisher","DOI":"10.3390\/s25071989","volume":"25","author":"B Zhang","year":"2025","unstructured":"B. Zhang, H. Cui, V. Nguyen, M. Whitty, Audio deepfake detection: what has been achieved and what lies ahead. Sensors 25(7), 1989 (2025)","journal-title":"Sensors"},{"key":"217_CR27","doi-asserted-by":"publisher","unstructured":"R.K. Bhukya, A. Raj, D.N. Raja, Audio deepfakes: feature extraction and model evaluation for detection. in 2024 5th International Conference for Emerging Technology (INCET), (IEEE,\u00a0Belgaum, 2024), pp. 1\u20136.\u00a0https:\/\/doi.org\/10.1109\/INCET61516.2024.10593405","DOI":"10.1109\/INCET61516.2024.10593405"},{"issue":"1","key":"217_CR28","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-025-93032-2","volume":"15","author":"AS Al-Shamayleh","year":"2025","unstructured":"A.S. Al-Shamayleh, H. Riasat, A.S. Alluhaidan, A. Raza, S.A. El-Rahman, D.S. AbdElminaam, Novel transfer learning based acoustic feature engineering for scene fake audio detection. Sci. Rep. 15(1), 8066 (2025)","journal-title":"Sci. Rep."},{"issue":"8","key":"217_CR29","doi-asserted-by":"publisher","DOI":"10.1111\/exsy.13322","volume":"40","author":"A Dixit","year":"2023","unstructured":"A. Dixit, N. Kaur, S. Kingra, Review of audio deepfake detection techniques: issues and prospects. Expert Syst. 40(8), e13322 (2023)","journal-title":"Expert Syst."},{"key":"217_CR30","doi-asserted-by":"publisher","first-page":"134018","DOI":"10.1109\/ACCESS.2022.3231480","volume":"10","author":"A Hamza","year":"2022","unstructured":"A. Hamza, A.R.R. Javed, F. Iqbal, N. Kryvinska, A.S. Almadhor, Z. Jalil, R. Borghol, Deepfake audio detection via MFCC features using machine learning. IEEE Access 10, 134018\u2013134028 (2022)","journal-title":"IEEE Access"},{"key":"217_CR31","doi-asserted-by":"publisher","unstructured":"M. Li, Y. Ahmadiadli, X.P. Zhang, A survey on speech deepfake detection. ACM Comput. Surv. 57(7), 1\u201338 (ACM, New York, 2025). https:\/\/doi.org\/10.1145\/3714458","DOI":"10.1145\/3714458"},{"key":"217_CR32","doi-asserted-by":"crossref","unstructured":"T.M. Wani, S.A.A. Qadri, D. Comminiello, I. Amerini, Detecting audio deepfakes: Integrating CNN and BiLSTM with multi-feature concatenation. in Proceedings of the 2024 ACM Workshop on Information Hiding and Multimedia Security, (Association for Computing Machinery,\u00a0New York, 2024), pp. 271\u2013276","DOI":"10.1145\/3658664.3659647"},{"key":"217_CR33","doi-asserted-by":"crossref","unstructured":"X. Zhang, J. Yi, C. Wang, C.Y. Zhang, S. Zeng, J. Tao, What to remember: Self-adaptive continual learning for audio deepfake detection. in Proceedings of the AAAI Conference on Artificial Intelligence, 38(17),\u00a019569\u201319577 (AAAI Press,\u00a0Washington, DC, 2024)\u00a0https:\/\/www.aaai.org\/Library\/AAAI\/aaai24contents.php","DOI":"10.1609\/aaai.v38i17.29929"},{"key":"217_CR34","doi-asserted-by":"publisher","unstructured":"F. Dong, Q. Tang, Y. Bai, Z. Wang, Advancing continual learning for robust deepfake audio classification.\u00a0Proceedings of the IEEE Region 10 Conference (TENCON 2024), p.\u00a0302\u2013305 (IEEE, 2024).\u00a0arXiv\u00a0preprint\u00a0arXiv:2407.10108.\u00a0https:\/\/doi.org\/10.1109\/TENCON61640.2024.10902761","DOI":"10.1109\/TENCON61640.2024.10902761"},{"key":"217_CR35","unstructured":"T.D.N. Le, K.K. Teh, H.D. Tran, Continuous learning of transformer-based audio deepfake detection.\u00a0(2024).\u00a0arXiv\u00a0preprint\u00a0arXiv:2409.05924.\u00a0https:\/\/arxiv.org\/abs\/2409.05924"},{"key":"217_CR36","doi-asserted-by":"publisher","unstructured":"K. Zaman, I.J. Samiul, M. Sah, C. Direkoglu, S. Okada, M. Unoki, Hybrid transformer architectures with diverse audio features for deepfake speech classification. IEEE Access. 12,\u00a0149221\u2013149237 (IEEE, 2024). https:\/\/doi.org\/10.1109\/ACCESS.2024.3478731","DOI":"10.1109\/ACCESS.2024.3478731"},{"issue":"2","key":"217_CR37","doi-asserted-by":"publisher","first-page":"252","DOI":"10.1109\/TBIOM.2021.3059479","volume":"3","author":"A Nautsch","year":"2021","unstructured":"A. Nautsch, X. Wang, N. Evans, T.H. Kinnunen, V. Vestman, M. Todisco, H. Delgado, M. Sahidullah, J. Yamagishi, K.A. Lee, Asvspoof 2019: spoofing countermeasures for the detection of synthesized, converted and replayed speech. IEEE Trans. Biom. Behav. Identity Sci. 3(2), 252\u2013265 (2021)","journal-title":"IEEE Trans. Biom. Behav. Identity Sci."},{"key":"217_CR38","doi-asserted-by":"publisher","unstructured":"R. Reimao, V. Tzerpos, For: A dataset for synthetic speech detection. in 2019 International Conference on Speech Technology and Human-Computer Dialogue (SpeD) (IEEE,\u00a0Timisoara, 2019), pp. 1\u201310.\u00a0https:\/\/doi.org\/10.1109\/SPED.2019.8906599","DOI":"10.1109\/SPED.2019.8906599"},{"key":"217_CR39","doi-asserted-by":"crossref","unstructured":"M. Todisco, X. Wang, V. Vestman, et. al.,\u00a0ASVspoof 2019: Future Horizons in Spoofed and Fake Audio Detection. Proceedings of the ISCA Annual Conference of the International Speech Communication Association (INTERSPEECH 2019),\u00a0p.\u00a01008\u20131012 (Graz, Austria 2019).\u00a0arXiv\u00a0preprint\u00a0arXiv:1904.05441.\u00a0https:\/\/www.isca-archive.org\/interspeech_2019\/todisco19_interspeech.html","DOI":"10.21437\/Interspeech.2019-2249"},{"key":"217_CR40","doi-asserted-by":"publisher","first-page":"2507","DOI":"10.1109\/TASLP.2023.3285283","volume":"31","author":"X Liu","year":"2023","unstructured":"X. Liu, X. Wang, M. Sahidullah, J. Patino, H. Delgado, T. Kinnunen, M. Todisco, J. Yamagishi, N. Evans, A. Nautsch et al., Asvspoof 2021: towards spoofed and deepfake speech detection in the wild. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 2507\u20132522 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"217_CR41","doi-asserted-by":"publisher","unstructured":"X. Wang, H. Delgado, H. Tak, J.w. Jung, H.j. Shim, M. Todisco, I. Kukanov, X. Liu, M. Sahidullah, T. Kinnunen et al.,\u00a0ASVspoof} 5: Crowdsourced Speech Data, Deepfakes, and Adversarial Attacks at Scale.\u00a0Proc. The Automatic Speaker Verification Spoofing Countermeasures Workshop (ASVspoof 2024), p.\u00a01\u20138, (Kos, Greece 2024).\u00a0arXiv\u00a0preprint\u00a0arXiv:2408.08739. https:\/\/doi.org\/10.21437\/ASVspoof.2024-1","DOI":"10.21437\/ASVspoof.2024-1"},{"key":"217_CR42","doi-asserted-by":"publisher","unstructured":"T.M. Wani, I. Amerini, Deepfakes audio detection leveraging audio spectrogram and convolutional neural networks. in International Conference on Image Analysis and Processing (Springer,\u00a0Cham, 2023), pp. 156\u2013167.\u00a0https:\/\/doi.org\/10.1007\/978-3-031-43153-1_14","DOI":"10.1007\/978-3-031-43153-1_14"},{"key":"217_CR43","doi-asserted-by":"crossref","unstructured":"T.M. Wani, R. Gulzar, I. Amerini, Abc-capsnet: Attention based cascaded capsule network for audio deepfake detection. in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (IEEE,\u00a0Seattle, 2024), pp. 2464\u20132472","DOI":"10.1109\/CVPRW63382.2024.00253"},{"key":"217_CR44","doi-asserted-by":"publisher","unstructured":"S. Modak, A.K. Das, R. Naskar, SpecViT: A Custom Vision-Transformer based Approach for Audio Deepfake Detection. in ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (IEEE,\u00a0Hyderabad, 2025), pp. 1\u20135.\u00a0https:\/\/doi.org\/10.1109\/ICASSP49660.2025.10889022","DOI":"10.1109\/ICASSP49660.2025.10889022"},{"key":"217_CR45","doi-asserted-by":"crossref","unstructured":"K. Zhang, Z. Hua, R. Lan, Y. Guo, Y. Zhang, G. Xu, Multi-View Collaborative Learning Network for Speech Deepfake Detection. in Proceedings of the AAAI Conference on Artificial Intelligence, 39(1), 1075\u20131083 (AAAI Press,\u00a0Philadelphia, 2025)","DOI":"10.1609\/aaai.v39i1.32094"},{"key":"217_CR46","doi-asserted-by":"crossref","unstructured":"K. Zhang, Z. Hua, R. Lan, Y. Zhang, Y. Guo, Phoneme-level feature discrepancies: a key to detecting sophisticated speech deepfakes.\u00a0Proceedings of the AAAI Conference on Artificial Intelligence. 39(1), 1066\u20131074 (AAAI Press,\u00a0Philadelphia, 2025)","DOI":"10.1609\/aaai.v39i1.32093"},{"key":"217_CR47","doi-asserted-by":"publisher","unstructured":"G. Ulutas, G. Tahaoglu, B. Ustubioglu, Deepfake Audio Detection with Vision Transformer Based Method. in 2023 46th International Conference on Telecommunications and Signal Processing (TSP) (IEEE,\u00a0Prague, 2023), pp. 244\u2013247.\u00a0https:\/\/doi.org\/10.1109\/TSP59544.2023.10197715","DOI":"10.1109\/TSP59544.2023.10197715"},{"key":"217_CR48","doi-asserted-by":"publisher","unstructured":"A.K.S. Yadav, Z. Xiang, K. Bhagtani, P. Bestagini, S. Tubaro, E.J. Delp, PS3DT: Synthetic Speech Detection Using Patched Spectrogram Transformer. in 2023 International Conference on Machine Learning and Applications (ICMLA) (IEEE,\u00a0Jacksonville, 2023), pp. 496\u2013503.\u00a0https:\/\/doi.org\/10.1109\/ICMLA58977.2023.00075","DOI":"10.1109\/ICMLA58977.2023.00075"},{"key":"217_CR49","doi-asserted-by":"publisher","unstructured":"L. Cuccovillo, M. Gerhardt, P. Aichroth, Audio Transformer for Synthetic Speech Detection via Multi-Formant Analysis. in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)\u00a0Workshops,\u00a0(IEEE,\u00a0Seattle, 2024), pp. 4409\u20134417. https:\/\/doi.org\/10.1109\/CVPR52733.2024.00001","DOI":"10.1109\/CVPR52733.2024.00001"},{"key":"217_CR50","doi-asserted-by":"publisher","unstructured":"X. Li, K. Li, Y. Zheng, C. Yan, X. Ji, W. Xu, Safeear: Content privacy-preserving audio deepfake detection. in Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security, (Association for Computing Machinery,\u00a0New York, 2024), pp. 3585\u20133599.\u00a0https:\/\/doi.org\/10.1145\/3658644.3670285","DOI":"10.1145\/3658644.3670285"},{"key":"217_CR51","first-page":"67901","volume":"37","author":"Y Zhu","year":"2024","unstructured":"Y. Zhu, S. Koppisetti, T. Tran, G. Bharaj, Slim: style-linguistics mismatch model for generalized audio deepfake detection. Adv. Neural. Inf. Process. Syst. 37, 67901\u201367928 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"217_CR52","unstructured":"Wiseman. Python interface to the webrtc voice activity detector. GitHub repository. 2016. https:\/\/github.com\/wiseman\/py-webrtcvad. Online. Accessed May-2025"},{"key":"217_CR53","first-page":"17283","volume":"33","author":"M Zaheer","year":"2020","unstructured":"M. Zaheer, G. Guruganesh, K.A. Dubey, J. Ainslie, C. Alberti, S. Ontanon, P. Pham, A. Ravula, Q. Wang, L. Yang et al., Big bird: transformers for longer sequences. Adv. Neural. Inf. Process. Syst. 33, 17283\u201317297 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"217_CR54","doi-asserted-by":"publisher","unstructured":"J. Frank, L. Sch\u00f6nherr, Wavefake: A data set to facilitate audio deepfake detection.\u00a0Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2), (2021).\u00a0arXiv\u00a0preprint\u00a0arXiv:2111.02813. https:\/\/doi.org\/10.48550\/arXiv.2111.02813","DOI":"10.48550\/arXiv.2111.02813"},{"key":"217_CR55","doi-asserted-by":"publisher","unstructured":"J.w. Jung, H.S. Heo, H. Tak, H.j. Shim, J.S. Chung, B.J. Lee, H.J. Yu, N. Evans, Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks. in ICASSP 2022- IEEE international conference on acoustics, speech and signal processing,\u00a0(IEEE,\u00a0Singapore, 2022), pp. 6367\u20136371.\u00a0https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747766","DOI":"10.1109\/ICASSP43922.2022.9747766"},{"key":"217_CR56","doi-asserted-by":"publisher","unstructured":"Y. Gong, Y.A. Chung, J. Glass, Ast: Audio spectrogram transformer.\u00a0Proc. Interspeech 2021, p.\u00a0571\u2013575, (Brno, Czechia 2021).\u00a0arXiv\u00a0preprint\u00a0arXiv:2104.01778. https:\/\/doi.org\/10.21437\/Interspeech.2021-698","DOI":"10.21437\/Interspeech.2021-698"},{"issue":"6","key":"217_CR57","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"S. Chen, C. Wang, Z. Chen, Y. Wu, S. Liu, Z. Chen, J. Li, N. Kanda, T. Yoshioka, X. Xiao et al., Wavlm: large-scale self-supervised pre-training for full stack speech processing. IEEE J. Sel. Top. Signal Process. 16(6), 1505\u20131518 (2022)","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"217_CR58","unstructured":"S.g. Lee, W. Ping, B. Ginsburg, B. Catanzaro, S. Yoon, Bigvgan: A universal neural vocoder with large-scale training.\u00a0The Eleventh International Conference on Learning Representations, (2022).\u00a0arXiv\u00a0preprint\u00a0arXiv:2206.04658.\u00a0https:\/\/openreview.net\/forum?id=iTtGCMDEzS"},{"key":"217_CR59","unstructured":"A. Garg, Z. Cai, L. Zhang, H.L. Xinyuan, L.P. Garc\u00eda-Perera, K. Duh, S. Khudanpur, M. Wiesner, N. Andrews, Shiftyspeech: A large-scale synthetic speech dataset with distribution shifts.\u00a0(2025).\u00a0arXiv\u00a0preprint\u00a0arXiv:2502.05674.\u00a0https:\/\/github.com\/Ashigarg123\/ShiftySpeech"}],"container-title":["EURASIP Journal on Information Security"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13635-025-00217-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13635-025-00217-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13635-025-00217-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T13:28:14Z","timestamp":1761744494000},"score":1,"resource":{"primary":{"URL":"https:\/\/jis-eurasipjournals.springeropen.com\/articles\/10.1186\/s13635-025-00217-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,29]]},"references-count":59,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["217"],"URL":"https:\/\/doi.org\/10.1186\/s13635-025-00217-3","relation":{},"ISSN":["2510-523X"],"issn-type":[{"value":"2510-523X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,29]]},"assertion":[{"value":"9 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"32"}}