{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T16:04:52Z","timestamp":1775837092554,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","funder":[{"DOI":"10.13039\/100006754","name":"Army Research Laboratory","doi-asserted-by":"publisher","award":["W911NF-23-2-0224"],"award-info":[{"award-number":["W911NF-23-2-0224"]}],"id":[{"id":"10.13039\/100006754","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,13]]},"DOI":"10.1145\/3716553.3750779","type":"proceedings-article","created":{"date-parts":[[2025,10,11]],"date-time":"2025-10-11T13:13:16Z","timestamp":1760188396000},"page":"424-433","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Unobtrusive Universal Acoustic Adversarial Attacks on Speech Foundation Models in the Wild"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-6658-3362","authenticated-orcid":false,"given":"Jayden","family":"Fassett","sequence":"first","affiliation":[{"name":"Computer Science, Georgia State University, Atlanta, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7559-2829","authenticated-orcid":false,"given":"Anjila","family":"Budathoki","sequence":"additional","affiliation":[{"name":"Computer Science, Georgia State University, Atlanta, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8360-8755","authenticated-orcid":false,"given":"Jack","family":"Morris","sequence":"additional","affiliation":[{"name":"Computer Science, Georgia State University, Atlanta, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8847-8345","authenticated-orcid":false,"given":"Qin","family":"Hu","sequence":"additional","affiliation":[{"name":"Computer Science, Georgia State University, Atlanta, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1811-3225","authenticated-orcid":false,"given":"Yi","family":"Ding","sequence":"additional","affiliation":[{"name":"Computer Science, Georgia State University, Atlanta, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,12]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3610661.3616129"},{"key":"e_1_3_3_2_3_2","unstructured":"Alexandre D\u00e9fossez Laurent Mazar\u00e9 Manu Orsini Am\u00e9lie Royer Patrick P\u00e9rez Herv\u00e9 J\u00e9gou Edouard Grave and Neil Zeghidour. 2024. Moshi: a speech-text foundation model for real-time dialogue. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.00037 (2024)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Yunjie Ge Lingchen Zhao Qian Wang Yiheng Duan and Minxin Du. 2023. Advddos: Zero-query adversarial attacks against commercial speech recognition systems. IEEE Transactions on Information Forensics and Security 18 (2023) 3647\u20133661.","DOI":"10.1109\/TIFS.2023.3283915"},{"key":"e_1_3_3_2_5_2","unstructured":"Ian\u00a0J. Goodfellow Jonathon Shlens and Christian Szegedy. 2015. Explaining and Harnessing Adversarial Examples. arxiv:https:\/\/arXiv.org\/abs\/1412.6572\u00a0[stat.ML] https:\/\/arxiv.org\/abs\/1412.6572"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-99579-321"},{"key":"e_1_3_3_2_7_2","unstructured":"Andrew Ilyas Shibani Santurkar Dimitris Tsipras Logan Engstrom Brandon Tran and Aleksander Madry. 2019. Adversarial Examples Are Not Bugs They Are Features. arxiv:https:\/\/arXiv.org\/abs\/1905.02175\u00a0[stat.ML] https:\/\/arxiv.org\/abs\/1905.02175"},{"key":"e_1_3_3_2_8_2","unstructured":"ISO. 2003. Acoustics\u2014Normal equal-loudness-level contours."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462693"},{"key":"e_1_3_3_2_10_2","unstructured":"Yinghao\u00a0Aaron Li Xilin Jiang Jordan Darefsky Ge Zhu and Nima Mesgarani. 2024. Style-talker: Finetuning audio language model and style-based text-to-speech model for fast spoken dialogue generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.11849 (2024)."},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-07974-5_2"},{"key":"e_1_3_3_2_12_2","unstructured":"Aleksander Madry Aleksandar Makelov Ludwig Schmidt Dimitris Tsipras and Adrian Vladu. 2017. Towards Deep Learning Models Resistant to Adversarial Attacks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1706.06083 (2017)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"J.\u00a0L. Mitchell. 2004. Introduction to Digital Audio Coding and Standards. Journal of Electronic Imaging 13 2 (2004) 399.","DOI":"10.1117\/1.1695413"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Brian\u00a0CJ Moore and Brian\u00a0R Glasberg. 1983. Suggested formulae for calculating auditory-filter bandwidths and excitation patterns. The Journal of the Acoustical Society of America 74 3 (1983) 750\u2013753.","DOI":"10.1121\/1.389861"},{"key":"e_1_3_3_2_15_2","unstructured":"Seyed-Mohsen Moosavi-Dezfooli Alhussein Fawzi Omar Fawzi and Pascal Frossard. 2017. Universal adversarial perturbations. arxiv:https:\/\/arXiv.org\/abs\/1610.08401\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1610.08401"},{"key":"e_1_3_3_2_16_2","unstructured":"Raphael Olivier and Bhiksha Raj. 2023. There is more than one kind of robustness: Fooling Whisper with adversarial examples. arxiv:https:\/\/arXiv.org\/abs\/2210.17316\u00a0[eess.AS] https:\/\/arxiv.org\/abs\/2210.17316"},{"key":"e_1_3_3_2_17_2","volume-title":"Discrete\u2011Time Signal Processing","author":"Oppenheim Alan\u00a0V","year":"1999","unstructured":"Alan\u00a0V Oppenheim and Ronald\u00a0W Schafer. 1999. Discrete\u2011Time Signal Processing. Prentice Hall."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_3_2_19_2","unstructured":"Yao Qin Nicholas Carlini Ian Goodfellow Garrison Cottrell and Colin Raffel. 2019. Imperceptible Robust and Targeted Adversarial Examples for Automatic Speech Recognition. arxiv:https:\/\/arXiv.org\/abs\/1903.10346\u00a0[eess.AS] https:\/\/arxiv.org\/abs\/1903.10346"},{"key":"e_1_3_3_2_20_2","unstructured":"Alec Radford Jong\u00a0Wook Kim Tao Xu Greg Brockman Christine McLeavey and Ilya Sutskever. 2022. Robust Speech Recognition via Large-Scale Weak Supervision. arxiv:https:\/\/arXiv.org\/abs\/2212.04356\u00a0[eess.AS] https:\/\/arxiv.org\/abs\/2212.04356"},{"key":"e_1_3_3_2_21_2","first-page":"28492","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492\u201328518."},{"key":"e_1_3_3_2_22_2","unstructured":"Vyas Raina and Mark Gales. 2024. Controlling Whisper: Universal Acoustic Adversarial Attacks to Control Speech Foundation Models. arxiv:https:\/\/arXiv.org\/abs\/2407.04482\u00a0[cs.SD] https:\/\/arxiv.org\/abs\/2407.04482"},{"key":"e_1_3_3_2_23_2","unstructured":"Vyas Raina Rao Ma Charles McGhee Kate Knill and Mark Gales. 2024. Muting Whisper: A Universal Acoustic Adversarial Attack on Speech Foundation Models. arxiv:https:\/\/arXiv.org\/abs\/2405.06134\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2405.06134"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-48309-7_46"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Andrew Rouditchenko Yuan Gong Samuel Thomas Leonid Karlinsky Hilde Kuehne Rogerio Feris and James Glass. 2024. Whisper-flamingo: Integrating visual features into whisper for audio-visual speech recognition and translation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.10082 (2024).","DOI":"10.21437\/Interspeech.2024-322"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","unstructured":"S.\u00a0S. Stevens J. Volkmann and E.\u00a0B. Newman. 1937. A scale for the measurement of the psychological magnitude pitch. Journal of the Acoustical Society of America 8 3 (1937) 185\u2013190. 10.1121\/1.1915893","DOI":"10.1121\/1.1915893"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446224"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1505"},{"key":"e_1_3_3_2_29_2","unstructured":"W.\u00a0A. K.\u00a0M. Wickramaarachchi Sameeri\u00a0Sathsara Subasinghe K.\u00a0K. Rashani\u00a0Tharushika Wijerathna A.\u00a0Sahashra\u00a0Udani Athukorala Lakmini Abeywardhana and A. Karunasena. 2024. Identifying False Content and Hate Speech in Sinhala YouTube Videos by Analyzing the Audio. arxiv:https:\/\/arXiv.org\/abs\/2402.01752\u00a0[eess.AS] https:\/\/arxiv.org\/abs\/2402.01752"},{"key":"e_1_3_3_2_30_2","unstructured":"Yao-Yuan Yang Moto Hira Zhaoheng Ni Anjali Chourdia Artyom Astafurov Caroline Chen Ching-Feng Yeh Christian Puhrsch David Pollack Dmitriy Genzel Donny Greenberg Edward\u00a0Z. Yang Jason Lian Jay Mahadeokar Jeff Hwang Ji Chen Peter Goldsborough Prabhat Roy Sean Narenthiran Shinji Watanabe Soumith Chintala Vincent Quenneville-B\u00e9lair and Yangyang Shi. 2021. TorchAudio: Building Blocks for Audio and Speech Processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2110.15018 (2021)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3686215.3688371"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","unstructured":"Eberhard Zwicker. 1961. Subdivision of the Audible Frequency Range into Critical Bands (Frequenzgruppen). The Journal of the Acoustical Society of America 33 2 (1961) 248\u2013248. 10.1121\/1.1908630","DOI":"10.1121\/1.1908630"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.5555\/1209342"}],"event":{"name":"ICMI '25: International Conference on Multimodal Interaction","location":"Canberra Australia","acronym":"ICMI '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 27th International Conference on Multimodal Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3716553.3750779","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T22:29:18Z","timestamp":1769466558000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3716553.3750779"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,12]]},"references-count":32,"alternative-id":["10.1145\/3716553.3750779","10.1145\/3716553"],"URL":"https:\/\/doi.org\/10.1145\/3716553.3750779","relation":{},"subject":[],"published":{"date-parts":[[2025,10,12]]},"assertion":[{"value":"2025-10-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}