{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T21:44:50Z","timestamp":1785707090985,"version":"3.56.0"},"reference-count":107,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T00:00:00Z","timestamp":1785283200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1016\/j.csl.2026.102037","type":"journal-article","created":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T06:37:14Z","timestamp":1785220634000},"page":"102037","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Robustness in deepfake speech detection: A survey of failure mechanisms including an experimental case study"],"prefix":"10.1016","volume":"102","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0142-6575","authenticated-orcid":false,"given":"Harry","family":"Maltby","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Julie","family":"Wall","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cornelius","family":"Glackin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mansour","family":"Moniri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Iwa","family":"Salami","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Letian","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nigel","family":"Cannings","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.102037_b1","series-title":"A review of modern audio deepfake detection methods: Challenges and future directions","author":"Almutairi","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b2","series-title":"Proceedings of the 23rd Annual Conference on Information Technology Education","article-title":"Availability of voice deepfake technology and its impact for good and evil","author":"Amezaga","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b3","series-title":"Combining automatic speaker verification and prosody analysis for synthetic speech detection","author":"Attorresi","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b4","series-title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale","author":"Babu","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b5","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.102037_b6","series-title":"International Symposium on Foundations and Practice of Security","first-page":"355","article-title":"Creation and detection of german voice deepfakes","author":"Barnekow","year":"2021"},{"issue":"1","key":"10.1016\/j.csl.2026.102037_b7","doi-asserted-by":"crossref","first-page":"tyad011","DOI":"10.1093\/cybsec\/tyad011","article-title":"Testing human ability to detect \u2018deepfake\u2019images of human faces","volume":"9","author":"Bray","year":"2023","journal-title":"J. Cybersecur."},{"key":"10.1016\/j.csl.2026.102037_b8","series-title":"Fraudsters cloned company director\u2019s voice in 35 million heist, police find","author":"Brewster","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b9","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.102037_b10","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Waveform boundary detection for partially spoofed audio","author":"Cai","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b11","first-page":"1","article-title":"Deep learning on computational-resource-limited platforms: A survey","volume":"2020","author":"Chen","year":"2020","journal-title":"Mob. Inf. Syst."},{"key":"10.1016\/j.csl.2026.102037_b12","series-title":"UR channel-robust synthetic speech detection system for ASVspoof 2021","author":"Chen","year":"2021"},{"issue":"5","key":"10.1016\/j.csl.2026.102037_b13","doi-asserted-by":"crossref","first-page":"1024","DOI":"10.1109\/JSTSP.2020.2999185","article-title":"Recurrent convolutional structures for audio spoof and video deepfake detection","volume":"14","author":"Chintha","year":"2020","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.csl.2026.102037_b14","doi-asserted-by":"crossref","first-page":"56","DOI":"10.1016\/j.specom.2022.04.005","article-title":"A study on data augmentation in voice anti-spoofing","volume":"141","author":"Cohen","year":"2022","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.102037_b15","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"8962","article-title":"Deepfake speech detection through emotion recognition: A semantic approach","author":"Conti","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b16","series-title":"A voice deepfake was used to scam a CEO out of $243,000","author":"Damiani","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b17","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Bts-e: Audio deepfake detection using breathing-talking-silence encoder","author":"Doan","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b18","series-title":"Adaptive re-calibration of channel-wise features for adversarial audio classification","author":"Dongre","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b19","series-title":"2023 International Conference on System, Computation, Automation and Networking","first-page":"1","article-title":"Audio deepfake detection using data augmented graph frequency cepstral coefficients","author":"Dua","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b20","first-page":"226","article-title":"A density-based algorithm for discovering clusters in large spatial databases with noise","volume":"vol. 96","author":"Ester","year":"1996"},{"key":"10.1016\/j.csl.2026.102037_b21","series-title":"Proceedings of INTERSPEECH","article-title":"Spoofing and countermeasures for automatic speaker verification","author":"Evans","year":"2013"},{"key":"10.1016\/j.csl.2026.102037_b22","series-title":"2022 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Mel-spectrogram image-based end-to-end audio deepfake detection under channel-mismatched conditions","author":"Fathan","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b23","doi-asserted-by":"crossref","unstructured":"Firc, A., Malinka, K., Han\u00e1\u010dek, P., 2024. Deepfake Speech Detection: A Spectrogram Analysis. In: Proceedings of the 39th ACM\/SIGAPP Symposium on Applied Computing. pp. 1312\u20131320.","DOI":"10.1145\/3605098.3635911"},{"key":"10.1016\/j.csl.2026.102037_b24","series-title":"Wavefake: A data set to facilitate audio deepfake detection","author":"Frank","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b25","series-title":"Perturbed public voices (p2 v): A dataset for robust audio deepfake detection","author":"Gao","year":"2025"},{"key":"10.1016\/j.csl.2026.102037_b26","series-title":"Explainable deepfake and spoofing detection: an attack analysis using shapley additive explanations","author":"Ge","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b27","series-title":"2022 IEEE 7th International Conference for Convergence in Technology (I2CT)","first-page":"1","article-title":"Detection of morphed face, body, audio signals using deep neural networks","author":"Gharde","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b28","series-title":"Cmkd: Cnn\/transformer-based cross-model knowledge distillation for audio classification","author":"Gong","year":"2022"},{"issue":"1","key":"10.1016\/j.csl.2026.102037_b29","doi-asserted-by":"crossref","DOI":"10.1073\/pnas.2110013119","article-title":"Deepfake detection by human crowds, machines, and machine-informed crowds","volume":"119","author":"Groh","year":"2022","journal-title":"Proc. Natl. Acad. Sci."},{"key":"10.1016\/j.csl.2026.102037_b30","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"12702","article-title":"Audio deepfake detection with self-supervised wavlm and multi-fusion attentive classifier","author":"Guo","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b31","series-title":"Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming","first-page":"1","article-title":"Beyond human-level accuracy: computational challenges in deep learning","author":"Hestness","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b32","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102037_b33","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G., 2018. Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 7132\u20137141.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"10.1016\/j.csl.2026.102037_b34","series-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b35","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"9985","article-title":"SpeechFake: A large-scale multilingual speech deepfake dataset incorporating cutting-edge generation methods","author":"Huang","year":"2025"},{"key":"10.1016\/j.csl.2026.102037_b36","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2022.116770","article-title":"Voice spoofing detector: A unified anti-spoofing framework","volume":"198","author":"Javed","year":"2022","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.csl.2026.102037_b37","series-title":"Rawnet: Advanced end-to-end deep neural network using raw waveforms for text-independent speaker verification","author":"Jung","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b38","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6367","article-title":"Aasist: Audio anti-spoofing using integrated spectro-temporal graph attention networks","author":"Jung","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b39","series-title":"Advancements of AI and machine learning in FinTech industry (2016\u20132020)","author":"Kamuangu","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b40","series-title":"Improved DeepFake detection using whisper features","author":"Kawa","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b41","series-title":"Attack agnostic dataset: Towards generalization and stabilization of audio DeepFake detection","author":"Kawa","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b42","series-title":"Deepfake audio detection","author":"Khan","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b43","series-title":"Voice spoofing countermeasures: Taxonomy, state-of-the-art, experimental analysis of generalizability, open challenges, and the way forward","author":"Khan","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b44","first-page":"1","article-title":"A deep learning framework for audio deepfake detection","author":"Khochare","year":"2021","journal-title":"Arab. J. Sci. Eng."},{"key":"10.1016\/j.csl.2026.102037_b45","series-title":"International Conference on Machine Learning","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","author":"Kim","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b46","series-title":"T-DCF: a detection cost function for the tandem assessment of spoofing countermeasures and automatic speaker verification","author":"Kinnunen","year":"2018"},{"key":"10.1016\/j.csl.2026.102037_b47","doi-asserted-by":"crossref","unstructured":"Kwak, I.Y., Choi, S., Yang, J., Lee, Y., Han, S., Oh, S., 2022. Low-Quality Fake Audio Detection through Frequency Feature Masking. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 9\u201317.","DOI":"10.1145\/3552466.3556533"},{"key":"10.1016\/j.csl.2026.102037_b48","series-title":"Dollars, Deception, and Deepfakes: An Analysis of Deepfakes and Synthetic Media Fraud","author":"Levine","year":"2020"},{"key":"10.1016\/j.csl.2026.102037_b49","doi-asserted-by":"crossref","unstructured":"Li, M., Ahmadiadli, Y., Zhang, X.-P., 2022. A Comparative Study on Physical and Perceptual Features for Deepfake Audio Detection. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 35\u201341.","DOI":"10.1145\/3552466.3556523"},{"key":"10.1016\/j.csl.2026.102037_b50","series-title":"A survey on speech deepfake detection","author":"Li","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b51","series-title":"International Conference on Machine Learning","first-page":"5958","article-title":"Train big, then compress: Rethinking model size for efficient training and inference of transformers","author":"Li","year":"2020"},{"key":"10.1016\/j.csl.2026.102037_b52","series-title":"2019 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2577","article-title":"\u201dHello? Who am I talking to?\u201d a shallow CNN approach for human vs. Bot speech classification","author":"Lieto","year":"2019"},{"issue":"8","key":"10.1016\/j.csl.2026.102037_b53","doi-asserted-by":"crossref","first-page":"3926","DOI":"10.3390\/app12083926","article-title":"Detecting deepfake voice using explainable deep learning techniques","volume":"12","author":"Lim","year":"2022","journal-title":"Appl. Sci."},{"key":"10.1016\/j.csl.2026.102037_b54","doi-asserted-by":"crossref","unstructured":"Liu, X., Liu, M., Zhang, L., Zhang, L., Zeng, C., Li, K.-C., Li, N., Lee, K.-A., Wang, L., Dang, J., 2022. Deep Spectro-temporal Artifacts for Detecting Synthesized Speech. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 69\u201375.","DOI":"10.1145\/3552466.3556527"},{"key":"10.1016\/j.csl.2026.102037_b55","doi-asserted-by":"crossref","first-page":"2507","DOI":"10.1109\/TASLP.2023.3285283","article-title":"Asvspoof 2021: Towards spoofed and deepfake speech detection in the wild","volume":"31","author":"Liu","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102037_b56","doi-asserted-by":"crossref","unstructured":"Lu, J., Li, Z., Zhang, Y., Wang, W., Zhang, P., 2022. Acoustic or Pattern? Speech Spoofing Countermeasure based on Image Pre-training Models. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 77\u201384.","DOI":"10.1145\/3552466.3556524"},{"key":"10.1016\/j.csl.2026.102037_b57","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9231","article-title":"Fake audio detection based on unsupervised pretraining models","author":"Lv","year":"2022"},{"issue":"8","key":"10.1016\/j.csl.2026.102037_b58","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pone.0285333","article-title":"Warning: Humans cannot reliably detect speech deepfakes","volume":"18","author":"Mai","year":"2023","journal-title":"Plos One"},{"key":"10.1016\/j.csl.2026.102037_b59","series-title":"2024 International Joint Conference on Neural Networks","first-page":"1","article-title":"A frequency bin analysis of distinctive ranges between human and deepfake generated voices","author":"Maltby","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b60","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9241","article-title":"The vicomtech audio deepfake detection system based on wav2vec2 for the 2022 ADD challenge","author":"Mart\u00edn-Do\u00f1as","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b61","series-title":"Umap: Uniform manifold approximation and projection for dimension reduction","author":"McInnes","year":"2018"},{"key":"10.1016\/j.csl.2026.102037_b62","series-title":"Does audio deepfake detection generalize?","author":"M\u00fcller","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b63","series-title":"Does audio deepfake detection generalize?","author":"M\u00fcller","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b64","series-title":"Speech is silver, silence is golden: What do ASVspoof-trained models really learn?","author":"M\u00fcller","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b65","series-title":"Attacker attribution of audio deepfakes","author":"M\u00fcller","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b66","series-title":"2024 International Joint Conference on Neural Networks","first-page":"1","article-title":"MLAAD: The multi-language audio anti-spoofing dataset","author":"M\u00fcller","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b67","series-title":"XLSR-MamBo: Scaling the hybrid mamba-attention backbone for audio deepfake detection","author":"Ng","year":"2026"},{"key":"10.1016\/j.csl.2026.102037_b68","series-title":"Specaugment: A simple data augmentation method for automatic speech recognition","author":"Park","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b69","series-title":"A comprehensive survey with critical analysis for deepfake speech detection","author":"Pham","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b70","series-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b71","series-title":"2019 International Conference on Speech Technology and Human-Computer Dialogue (SpeD)","first-page":"1","article-title":"For: A dataset for synthetic speech detection","author":"Reimao","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b72","doi-asserted-by":"crossref","first-page":"50851","DOI":"10.1109\/ACCESS.2023.3276480","article-title":"TIMIT-TTS: A text-to-speech dataset for multimodal synthetic media detection","volume":"11","author":"Salvi","year":"2023","journal-title":"IEEE Access"},{"key":"10.1016\/j.csl.2026.102037_b73","series-title":"Wav2vec: Unsupervised pre-training for speech recognition","author":"Schneider","year":"2019"},{"issue":"4","key":"10.1016\/j.csl.2026.102037_b74","doi-asserted-by":"crossref","first-page":"206","DOI":"10.1109\/11.362938","article-title":"Guide to MPEG-1 audio standard","volume":"40","author":"Shlien","year":"1994","journal-title":"IEEE Trans. Broadcast."},{"key":"10.1016\/j.csl.2026.102037_b75","doi-asserted-by":"crossref","unstructured":"Sun, C., Jia, S., Hou, S., Lyu, S., 2023. Ai-synthesized voice detection using neural vocoder artifacts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 904\u2013912.","DOI":"10.1109\/CVPRW59228.2023.00097"},{"key":"10.1016\/j.csl.2026.102037_b76","series-title":"ASVspoof 2019: Future horizons in spoofed and fake audio detection","author":"Todisco","year":"2019"},{"key":"10.1016\/j.csl.2026.102037_b77","series-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b78","article-title":"Deepfake speech detection: approaches from acoustic features related to auditory perception to deep neural networks","author":"UNOKI","year":"2024","journal-title":"IEICE Trans. Inf. Syst."},{"key":"10.1016\/j.csl.2026.102037_b79","series-title":"Backpropagation","first-page":"35","article-title":"Phoneme recognition using time-delay neural networks","author":"Waibel","year":"2013"},{"key":"10.1016\/j.csl.2026.102037_b80","series-title":"ASVspoof 5: Crowdsourced speech data, deepfakes, and adversarial attacks at scale","author":"Wang","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b81","series-title":"Investigating self-supervised front ends for speech spoofing countermeasures","author":"Wang","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b82","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2020.101114","article-title":"ASVspoof 2019: A large-scale public database of synthesized, converted and replayed speech","volume":"64","author":"Wang","year":"2020","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.102037_b83","doi-asserted-by":"crossref","unstructured":"Wang, C., Yi, J., Tao, J., Sun, H., Chen, X., Tian, Z., Ma, H., Fan, C., Fu, R., 2022a. Fully Automated End-to-End Fake Audio Detection. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 27\u201333.","DOI":"10.1145\/3552466.3556530"},{"key":"10.1016\/j.csl.2026.102037_b84","series-title":"Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia","first-page":"27","article-title":"Fully automated end-to-end fake audio detection","author":"Wang","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b85","first-page":"192","article-title":"Deepfake audio detection: a deep learning-based solution for group conversations","volume":"vol. 1","author":"Wijethunga","year":"2020"},{"key":"10.1016\/j.csl.2026.102037_b86","series-title":"Interspeech","article-title":"Spoofing speech detection by modeling local spectro-temporal and long-term dependency.","author":"Wu","year":"2024"},{"key":"10.1016\/j.csl.2026.102037_b87","series-title":"2022 5th International Conference on Pattern Recognition and Artificial Intelligence","first-page":"651","article-title":"Attentional fusion TDNN for spoof speech detection","author":"Wu","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b88","doi-asserted-by":"crossref","unstructured":"Wu, Z., Kinnunen, T., Evans, N., Yamagishi, J., Hanil\u00e7i, C., Sahidullah, M., Sizov, A., 2015. ASVspoof 2015: The First Automatic Speaker Verification Spoofing and Countermeasures Challenge. In: Sixteenth Annual Conference of the International Speech Communication Association.","DOI":"10.21437\/Interspeech.2015-462"},{"issue":"10","key":"10.1016\/j.csl.2026.102037_b89","doi-asserted-by":"crossref","first-page":"2609","DOI":"10.3390\/rs15102609","article-title":"Distillation sparsity training algorithm for accelerating convolutional neural networks in embedded systems","volume":"15","author":"Xiao","year":"2023","journal-title":"Remote. Sens."},{"key":"10.1016\/j.csl.2026.102037_b90","series-title":"Fake-mamba: Real-time speech deepfake detection using bidirectional mamba as self-attention\u2019s alternative","author":"Xuan","year":"2025"},{"key":"10.1016\/j.csl.2026.102037_b91","doi-asserted-by":"crossref","unstructured":"Xue, J., Fan, C., Lv, Z., Tao, J., Yi, J., Zheng, C., Wen, Z., Yuan, M., Shao, S., 2022. Audio deepfake detection based on a combination of f0 information and real plus imaginary spectrogram features. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 19\u201326.","DOI":"10.1145\/3552466.3556526"},{"key":"10.1016\/j.csl.2026.102037_b92","article-title":"Asvspoof 2019: Automatic speaker verification spoofing and countermeasures challenge evaluation plan","volume":"13","author":"Yamagishi","year":"2019","journal-title":"ASV Spoof"},{"key":"10.1016\/j.csl.2026.102037_b93","article-title":"Cstr vctk corpus: English multi-speaker corpus for cstr voice cloning toolkit (version 0.92)","author":"Yamagishi","year":"2019","journal-title":"Univ. Edinb. Cent. Speech Technol. Res. (CSTR)"},{"key":"10.1016\/j.csl.2026.102037_b94","series-title":"ASVspoof 2021: accelerating progress in spoofed and deepfake speech detection","author":"Yamagishi","year":"2021"},{"key":"10.1016\/j.csl.2026.102037_b95","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9226","article-title":"Audio deepfake detection system with neural stitching for add 2022","author":"Yan","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b96","series-title":"System fingerprints detection for DeepFake audio: An initial dataset and investigation","author":"Yan","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b97","doi-asserted-by":"crossref","first-page":"2160","DOI":"10.1109\/TIFS.2019.2956589","article-title":"Significance of subband features for synthetic speech detection","volume":"15","author":"Yang","year":"2020","journal-title":"IEEE Trans. Inf. Forensics Secur."},{"key":"10.1016\/j.csl.2026.102037_b98","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"9216","article-title":"Add 2022: the first audio deep synthesis detection challenge","author":"Yi","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b99","series-title":"Audio deepfake detection: A survey","author":"Yi","year":"2023"},{"key":"10.1016\/j.csl.2026.102037_b100","series-title":"ADD 2023: Towards audio deepfake detection and analysis in the wild","author":"Yi","year":"2024"},{"issue":"3","key":"10.1016\/j.csl.2026.102037_b101","doi-asserted-by":"crossref","first-page":"194","DOI":"10.3390\/info16030194","article-title":"A spoofing speech detection method combining multi-scale features and cross-layer information","volume":"16","author":"Yuan","year":"2025","journal-title":"Information"},{"issue":"7","key":"10.1016\/j.csl.2026.102037_b102","doi-asserted-by":"crossref","first-page":"1989","DOI":"10.3390\/s25071989","article-title":"Audio deepfake detection: What has been achieved and what lies ahead","volume":"25","author":"Zhang","year":"2025","journal-title":"Sensors (Basel, Switzerland)"},{"key":"10.1016\/j.csl.2026.102037_b103","doi-asserted-by":"crossref","first-page":"3374","DOI":"10.1109\/TASLP.2023.3306711","article-title":"The impact of silence on speech anti-spoofing","volume":"31","author":"Zhang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102037_b104","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, J., Wang, X., Li, Z., Xiao, R., Wang, W., Li, M., Zhang, P., 2022a. Deepfake Detection System for the ADD Challenge Track 3.2 Based on Score Fusion. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia. pp. 43\u201352.","DOI":"10.1145\/3552466.3556528"},{"issue":"6","key":"10.1016\/j.csl.2026.102037_b105","doi-asserted-by":"crossref","first-page":"1519","DOI":"10.1109\/JSTSP.2022.3182537","article-title":"BigSSL: Exploring the frontier of large-scale semi-supervised learning for automatic speech recognition","volume":"16","author":"Zhang","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.csl.2026.102037_b106","series-title":"International Conference on Neuroinformatics","first-page":"107","article-title":"Low-bit quantization of transformer for audio speech recognition","author":"Zharikov","year":"2022"},{"key":"10.1016\/j.csl.2026.102037_b107","series-title":"2021 7th International Conference on Signal Processing and Intelligent Systems","first-page":"1","article-title":"A countermeasure based on CQT spectrogram for deepfake speech detection","author":"Ziabary","year":"2021"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826001002?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826001002?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T20:50:06Z","timestamp":1785703806000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826001002"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,2]]},"references-count":107,"alternative-id":["S0885230826001002"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.102037","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2027,2]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Robustness in deepfake speech detection: A survey of failure mechanisms including an experimental case study","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.102037","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"102037"}}