{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:38:35Z","timestamp":1783193915439,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key R&D Program of China","award":["2023YFE0209800"],"award-info":[{"award-number":["2023YFE0209800"]}]},{"name":"NSFC","award":["U20B2049, U21B2018, 62302344, 62132011, 62161160337"],"award-info":[{"award-number":["U20B2049, U21B2018, 62302344, 62132011, 62161160337"]}]},{"name":"Shaanxi Province Key Industry Innovation Program","award":["2021ZDLGY01-02"],"award-info":[{"award-number":["2021ZDLGY01-02"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1145\/3658644.3670309","type":"proceedings-article","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T12:19:20Z","timestamp":1733746760000},"page":"630-644","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":16,"title":["Zero-Query Adversarial Attack on Black-box Automatic Speech Recognition Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9308-7452","authenticated-orcid":false,"given":"Zheng","family":"Fang","sequence":"first","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5265-1521","authenticated-orcid":false,"given":"Tao","family":"Wang","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1700-3836","authenticated-orcid":false,"given":"Lingchen","family":"Zhao","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2140-7131","authenticated-orcid":false,"given":"Shenyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5445-8209","authenticated-orcid":false,"given":"Bowen","family":"Li","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6158-3180","authenticated-orcid":false,"given":"Yunjie","family":"Ge","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8776-8730","authenticated-orcid":false,"given":"Qi","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6959-0569","authenticated-orcid":false,"given":"Chao","family":"Shen","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8967-8525","authenticated-orcid":false,"given":"Qian","family":"Wang","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,9]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2339736"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jsv.2016.10.043"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP40001.2021.00014"},{"key":"e_1_3_2_1_4_1","unstructured":"Alibaba. 2023. Alibaba Cloud Intelligent Speech Interaction. https:\/\/www.alibabacloud.com\/product\/intelligent-speech-interaction."},{"key":"e_1_3_2_1_5_1","unstructured":"Amazon. 2023. Alexa. https:\/\/www.alexa.com\/."},{"key":"e_1_3_2_1_6_1","volume-title":"Proc. of ICML. 173--182","author":"Amodei Dario","year":"2016","unstructured":"Dario Amodei, Sundaram Ananthanarayanan, Rishita Anubhai, Jingliang Bai, Eric Battenberg, Carl Case, Jared Casper, Bryan Catanzaro, Qiang Cheng, Guoliang Chen, et al. 2016. Deep speech 2: End-to-end speech recognition in english and mandarin. In Proc. of ICML. 173--182."},{"key":"e_1_3_2_1_7_1","unstructured":"Apple. 2023. Siri. https:\/\/www.apple.com\/siri\/."},{"key":"e_1_3_2_1_8_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems, Vol. 33 (2020), 12449--12460."},{"key":"e_1_3_2_1_9_1","volume-title":"Adversarial patch. CoRR","author":"Brown Tom B","year":"2017","unstructured":"Tom B Brown, Dandelion Man\u00e9, Aurko Roy, Mart\u00edn Abadi, and Justin Gilmer. 2017. Adversarial patch. CoRR, Vol. abs\/1712.09665 (2017)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/SPW.2018.00009"},{"key":"e_1_3_2_1_11_1","volume-title":"Proc. of USENIX Security. 2667--2684","author":"Chen Yuxuan","year":"2020","unstructured":"Yuxuan Chen, Xuejing Yuan, Jiangshan Zhang, Yue Zhao, Shengzhi Zhang, Kai Chen, and XiaoFeng Wang. 2020. Devil's Whisper: A general approach for physical adversarial attacks against commercial black-box speech recognition devices.. In Proc. of USENIX Security. 2667--2684."},{"key":"e_1_3_2_1_12_1","volume-title":"Proc. of USENIX Security. 321--338","author":"Demontis Ambra","year":"2019","unstructured":"Ambra Demontis, Marco Melis, Maura Pintor, Matthew Jagielski, Battista Biggio, Alina Oprea, Cristina Nita-Rotaru, and Fabio Roli. 2019. Why do adversarial attacks transfer? explaining transferability of evasion and poisoning attacks. In Proc. of USENIX Security. 321--338."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00957"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3548606.3559350"},{"key":"e_1_3_2_1_16_1","volume-title":"Zero-query adversarial attack on black-box automatic speech recognition systems. CoRR","author":"Fang Zheng","year":"1931","unstructured":"Zheng Fang, Tao Wang, Lingchen Zhao, Shenyi Zhang, Bowen Li, Yunjie Ge, Qi Li, Chao Shen, and Qian Wang. 2024. Zero-query adversarial attack on black-box automatic speech recognition systems. CoRR, Vol. abs\/2406.19311 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Foundations and Trends\u00ae in Signal Processing","volume":"1","author":"Gales Mark","year":"2008","unstructured":"Mark Gales, Steve Young, et al. 2008. The application of hidden Markov models in speech recognition. Foundations and Trends\u00ae in Signal Processing, Vol. 1 (2008), 195--304."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/SPW.2018.00016"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1201\/9781315154718"},{"key":"e_1_3_2_1_20_1","volume-title":"Sequence transduction with recurrent neural networks. CoRR","author":"Graves Alex","year":"2012","unstructured":"Alex Graves. 2012. Sequence transduction with recurrent neural networks. CoRR, Vol. abs\/1211.3711 (2012)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3548606.3560660"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2059"},{"key":"e_1_3_2_1_26_1","unstructured":"Awni Hannun Carl Case Jared Casper Bryan Catanzaro Greg Diamos Erich Elsen Ryan Prenger Sanjeev Satheesh Shubho Sengupta Adam Coates et al. 2014. Deep speech: Scaling up end-to-end speech recognition. CoRR Vol. abs\/1412.5567 (2014)."},{"key":"e_1_3_2_1_27_1","volume-title":"The CMA evolution strategy: A tutorial. CoRR","author":"Hansen Nikolaus","year":"2016","unstructured":"Nikolaus Hansen. 2016. The CMA evolution strategy: A tutorial. CoRR, Vol. abs\/1604.00772 (2016)."},{"key":"e_1_3_2_1_28_1","volume-title":"Perceptual linear predictive (PLP) analysis of speech. the Journal of the Acoustical Society of America","author":"Hermansky Hynek","year":"1990","unstructured":"Hynek Hermansky. 1990. Perceptual linear predictive (PLP) analysis of speech. the Journal of the Acoustical Society of America, Vol. 87 (1990), 1738--1752."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2404"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00483"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.1995189"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003750"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1938"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053889"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1819"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2846"},{"key":"e_1_3_2_1_38_1","volume-title":"Proc. of ICLR.","author":"Liu Yanpei","year":"2017","unstructured":"Yanpei Liu, Xinyun Chen, Chang Liu, and Dawn Song. 2017. Delving into transferable adversarial examples and black-box attacks. In Proc. of ICLR."},{"key":"e_1_3_2_1_39_1","volume-title":"Citrinet: Closing the gap between non-autoregressive and autoregressive end-to-end models for automatic speech recognition. CoRR","author":"Majumdar Somshubra","year":"2021","unstructured":"Somshubra Majumdar, Jagadeesh Balam, Oleksii Hrinchuk, Vitaly Lavrukhin, Vahid Noroozi, and Boris Ginsburg. 2021. Citrinet: Closing the gap between non-autoregressive and autoregressive end-to-end models for automatic speech recognition. CoRR, Vol. abs\/2104.01721 (2021)."},{"key":"e_1_3_2_1_40_1","unstructured":"Microsoft. 2023. Microsoft Azure Speech Service. https:\/\/azure.microsoft.com\/en-us\/products\/cognitive-services\/speech-services\/."},{"key":"e_1_3_2_1_41_1","unstructured":"Nvidia. 2023. NeMo. https:\/\/developer.nvidia.com\/nemo\/."},{"key":"e_1_3_2_1_42_1","unstructured":"OpenAI. 2023. Whisper. https:\/\/openai.com\/research\/whisper."},{"key":"e_1_3_2_1_43_1","unstructured":"OpenAI. 2023. Whisper large-v3. https:\/\/github.com\/openai\/whisper\/."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.5555\/3013545.3013549"},{"key":"e_1_3_2_1_45_1","volume-title":"Goodfellow","author":"Papernot Nicolas","year":"2016","unstructured":"Nicolas Papernot, Patrick D. McDaniel, and Ian J. Goodfellow. 2016. Transferability in machine learning: from phenomena to black-box attacks using adversarial samples. CoRR, Vol. abs\/1605.07277 (2016)."},{"key":"e_1_3_2_1_46_1","volume-title":"Proc. of NeruIPS. 8024--8035","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas Kopf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An imperative style, high-performance deep learning library. In Proc. of NeruIPS. 8024--8035."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096873"},{"key":"e_1_3_2_1_48_1","volume-title":"Proc. of ICML. 5231--5240","author":"Qin Yao","year":"2019","unstructured":"Yao Qin, Nicholas Carlini, Garrison Cottrell, Ian Goodfellow, and Colin Raffel. 2019. Imperceptible, robust, and targeted adversarial examples for automatic speech recognition. In Proc. of ICML. 5231--5240."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/5.18626"},{"key":"e_1_3_2_1_50_1","volume-title":"Proc. of ICML. 28492--28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In Proc. of ICML. 28492--28518."},{"key":"e_1_3_2_1_51_1","volume-title":"TREND: Transferability based robust ensemble design","author":"Ravikumar Deepak","year":"2022","unstructured":"Deepak Ravikumar, Sangamesh D Kodge, Isha Garg, and Kaushik Roy. 2022. TREND: Transferability based robust ensemble design. IEEE Transactions on Artificial Intelligence (2022)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2019.23288"},{"key":"e_1_3_2_1_54_1","volume-title":"Kaare Brandt Petersen, and Tue Lehn-Schi\u00f8ler","author":"Sigurdsson Sigurdur","year":"2006","unstructured":"Sigurdur Sigurdsson, Kaare Brandt Petersen, and Tue Lehn-Schi\u00f8ler. 2006. Mel Frequency Cepstral Coefficients: An Evaluation of Robustness of MP3 Encoded Music.. In ISMIR. 286--289."},{"key":"e_1_3_2_1_55_1","volume-title":"Proc. of ICLR.","author":"Szegedy Christian","year":"2014","unstructured":"Christian Szegedy, Wojciech Zaremba, Ilya Sutskever, Joan Bruna, Dumitru Erhan, Ian J. Goodfellow, and Rob Fergus. 2014. Intriguing properties of neural networks. In Proc. of ICLR."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/SPW.2019.00016"},{"key":"e_1_3_2_1_57_1","unstructured":"Tencent. 2023. Tencent Cloud Automatic Speech Recognition. https:\/\/cloud.tencent.com\/document\/product\/1093\/."},{"key":"e_1_3_2_1_58_1","volume-title":"Proc. of NeruIPS. 5998--6008","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Proc. of NeruIPS. 5998--6008."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2020.3026543"},{"key":"e_1_3_2_1_60_1","unstructured":"Lei Wu Zhanxing Zhu Cheng Tai and Weinan E. 2018. Understanding and enhancing the transferability of adversarial examples. CoRR Vol. abs\/1802.09707 (2018)."},{"key":"e_1_3_2_1_61_1","volume-title":"Proc. of USENIX Security. 247--264","author":"Wu Xinghui","year":"2023","unstructured":"Xinghui Wu, Shiqing Ma, Chao Shen, Chenhao Lin, Qian Wang, Qi Li, and Yuan Rao. 2023. KENKU: Towards efficient and stealthy black-box adversarial attacks against ASR systems. In Proc. of USENIX Security. 247--264."},{"key":"e_1_3_2_1_62_1","volume-title":"Proc. of ICLR.","author":"Yang Zhuolin","year":"2019","unstructured":"Zhuolin Yang, Bo Li, Pin-Yu Chen, and Dawn Song. 2019. Characterizing audio adversarial examples using temporal dependency. In Proc. of ICLR."},{"key":"e_1_3_2_1_63_1","volume-title":"Proc. of USENIX security. 3799--3816","author":"Yu Zhiyuan","year":"2023","unstructured":"Zhiyuan Yu, Yuanhaur Chang, Ning Zhang, and Chaowei Xiao. 2023. SMACK: Semantically meaningful adversarial audio attack. In Proc. of USENIX security. 3799--3816."},{"key":"e_1_3_2_1_64_1","volume-title":"Gunter","author":"Yuan Xuejing","year":"2018","unstructured":"Xuejing Yuan, Yuxuan Chen, Yue Zhao, Yunhui Long, Xiaokang Liu, Kai Chen, Shengzhi Zhang, Heqing Huang, XiaoFeng Wang, and Carl A. Gunter. 2018. CommanderSong: A systematic approach for practical adversarial voice recognition. In Proc. of USENIX Security. 49--64."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/DSN.2019.00019"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9004025"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053896"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460120.3485383"}],"event":{"name":"CCS '24: ACM SIGSAC Conference on Computer and Communications Security","location":"Salt Lake City UT USA","acronym":"CCS '24","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3670309","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658644.3670309","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:18:29Z","timestamp":1755843509000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3670309"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":68,"alternative-id":["10.1145\/3658644.3670309","10.1145\/3658644"],"URL":"https:\/\/doi.org\/10.1145\/3658644.3670309","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2024-12-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}