{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T23:04:11Z","timestamp":1784847851660,"version":"3.55.0"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,10,31]],"date-time":"2021-10-31T00:00:00Z","timestamp":1635638400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,10,31]],"date-time":"2021-10-31T00:00:00Z","timestamp":1635638400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006602","name":"Air Force Research Laboratory","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006602","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,10,31]]},"DOI":"10.1109\/ieeeconf53345.2021.9723142","type":"proceedings-article","created":{"date-parts":[[2022,3,4]],"date-time":"2022-03-04T20:26:46Z","timestamp":1646425606000},"page":"1426-1430","source":"Crossref","is-referenced-by-count":17,"title":["Synthesized Speech Detection Using Convolutional Transformer-Based Spectrogram Analysis"],"prefix":"10.1109","author":[{"given":"Emily R.","family":"Bartusiak","sequence":"first","affiliation":[{"name":"Purdue University,Video and Image Processing Lab (VIPER) School of Electrical and Computer Engineering,West Lafayette,IN,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Edward J.","family":"Delp","sequence":"additional","affiliation":[{"name":"Purdue University,Video and Image Processing Lab (VIPER) School of Electrical and Computer Engineering,West Lafayette,IN,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","article-title":"We Need No Pixels: Video Manipulation Detection Using Stream Descriptors","author":"g\u00fcera","year":"2019","journal-title":"Proceedings of the International Conference on Machine Learning Synthetic-Realities Deep Learning for Detecting AudioVisual Fakes Workshop"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/MIPR.2019.00024"},{"key":"ref33","first-page":"205","article-title":"Probabilistic Discriminative Models","author":"bishop","year":"2005","journal-title":"Pattern Recognition and Machine Learning"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/BF00994018"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1967.1053964"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.2307\/1403796"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/AVSS.2018.8639163"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00342"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143874"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.aci.2018.08.003"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICSPCS.2018.8631752"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/DSAA.2015.7344793"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/IWSSIP.2015.7314180"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1085"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462644"},{"key":"ref15","first-page":"209","article-title":"The Quefrency Alanysis of Time Series for Echoes: Cepstrum, Pseudo Autocovariance, Cross-Cepstrum and Saphe Cracking","volume":"15","author":"bogert","year":"1963","journal-title":"Proceedings of the Symposium on Time Series Analysis"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1121\/1.1915893"},{"key":"ref17","article-title":"Neuroscience","author":"purves","year":"2001","journal-title":"The Audible Spectrum"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2011.11.004"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2020.2999185"},{"key":"ref28","article-title":"Generative Adversarial Nets","volume":"27","author":"goodfellow","year":"2014","journal-title":"Proceedings of the conference on Neural Information Processing Systems"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00009"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref3","article-title":"Goldman Sachs, Ozy Media and a $40 Million Conference Call Gone Wrong","author":"smith","year":"2021","journal-title":"The New York Times"},{"key":"ref6","article-title":"The State of Deepfakes: Landscape, Threats, and Impact","author":"ajder","year":"2019","journal-title":"Deeptrace Lab"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.2352\/ISSN.2470-1173.2021.4.MWSF-273"},{"key":"ref5","article-title":"Deepfakes are Going to Wreck Havoc on Society. We are Not Prepared","author":"toews","year":"2020","journal-title":"Forbes"},{"key":"ref8","article-title":"Neural Style Transfer for Audio Spectrograms","author":"verma","year":"2017","journal-title":"Proceedings of the Conference on Neural Information Processing Systems Workshop for Machine Learning for Creativity and Design"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-01562-9"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TEM.2021.3117884"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2010.2100380"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1080\/02763869.2018.1404391"},{"key":"ref20","article-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","author":"dosovitskiy","year":"2021","journal-title":"Proceedings of the International Conference on Learning Representations"},{"key":"ref22","article-title":"Attention is All You Need","author":"vaswani","year":"2017","journal-title":"Proceedings of the Neural Information Processing Systems"},{"key":"ref21","article-title":"MLP-Mixer: An all-MLP Architecture for Vision","author":"tolstikhin","year":"2021"},{"key":"ref24","first-page":"4171","article-title":"BERT: Pretraining of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics Human Language Technologies Volume 1 (Long and Short Papers)"},{"key":"ref23","article-title":"Escaping the Big Data Paradigm with Compact Transformers","author":"hassani","year":"2021"},{"key":"ref26","article-title":"ASVspoof 2019: Automatic Speaker Verification Spoofing and Countermeasures Challenge Evaluation Plan","author":"todisco","year":"2019","journal-title":"ASVspoof Consortium"},{"key":"ref25","article-title":"ASVspoof 2019: The 3rd Automatic Speaker Verification Spoofing and Countermeasures Challenge database","author":"yamagishi","year":"2019","journal-title":"Centre for Speech Technology Research University of Edinburgh"}],"event":{"name":"2021 55th Asilomar Conference on Signals, Systems, and Computers","location":"Pacific Grove, CA, USA","start":{"date-parts":[[2021,10,31]]},"end":{"date-parts":[[2021,11,3]]}},"container-title":["2021 55th Asilomar Conference on Signals, Systems, and Computers"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9723034\/9723086\/09723142.pdf?arnumber=9723142","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,15]],"date-time":"2022-06-15T20:15:38Z","timestamp":1655324138000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9723142\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,31]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/ieeeconf53345.2021.9723142","relation":{},"subject":[],"published":{"date-parts":[[2021,10,31]]}}}