{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T11:19:11Z","timestamp":1780399151623,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819200733","type":"print"},{"value":"9789819200740","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-92-0074-0_27","type":"book-chapter","created":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T10:52:48Z","timestamp":1780397568000},"page":"381-397","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Speech-EA: Evolutionary Algorithm-Based Attack on\u00a0Automatic Speech Recognition Systems"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8292-8747","authenticated-orcid":false,"given":"Elmir","family":"Avdusinovic","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0141-4742","authenticated-orcid":false,"given":"Ali Osman","family":"Topal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8562-1433","authenticated-orcid":false,"given":"Enea","family":"Mancellari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7335-7682","authenticated-orcid":false,"given":"Volker","family":"M\u00fcller","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0241-9935","authenticated-orcid":false,"given":"Faraz Mohammad Mushtak","family":"Mogal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8808-2730","authenticated-orcid":false,"given":"Franck","family":"Lepr\u00e9vost","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,1]]},"reference":[{"key":"27_CR1","doi-asserted-by":"publisher","unstructured":"An, F., Wu, Q., Liu, B.: Audio Compensation with cascade biquad filters for feedback active noise control headphones. Processes 10(4) (2022). https:\/\/doi.org\/10.3390\/pr10040730","DOI":"10.3390\/pr10040730"},{"key":"27_CR2","unstructured":"Ava, T.: Live Professional Captions vs Live Automatic Captions (ASR). https:\/\/www.ava.me\/blog\/live-professional-captions-vs-live-automatic-captions-asr"},{"key":"27_CR3","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. ArXiv abs\/2006.11477 (2020). https:\/\/api.semanticscholar.org\/CorpusID:219966759"},{"key":"27_CR4","unstructured":"Buchner, J.: Synthetic speech commands: a public dataset for single-word speech recognition. (2017). https:\/\/www.kaggle.com\/jbuchner\/synthetic-speech-commands-dataset\/"},{"key":"27_CR5","doi-asserted-by":"publisher","unstructured":"Carlini, N., Wagner, D.: Audio adversarial examples: targeted attacks on speech-to-text. In: 2018 IEEE Security and Privacy Workshops (SPW), pp.\u00a01\u20137 (2018). https:\/\/doi.org\/10.1109\/SPW.2018.00009","DOI":"10.1109\/SPW.2018.00009"},{"key":"27_CR6","doi-asserted-by":"publisher","unstructured":"Chen, G., et al.: Who is Real Bob? Adversarial Attacks on Speaker Recognition Systems. In: 2021 IEEE Symposium on Security and Privacy (SP), pp. 694\u2013711. IEEE (2021). https:\/\/doi.org\/10.1109\/SP40001.2021.00004","DOI":"10.1109\/SP40001.2021.00004"},{"key":"27_CR7","unstructured":"Chen, Y., et al.: Devil\u2019s whisper: A general approach for physical adversarial attacks against commercial black-box speech recognition devices. In: 29th USENIX Security Symposium (USENIX Security 20), pp. 2667\u20132684. USENIX Association (2020). https:\/\/www.usenix.org\/conference\/usenixsecurity20\/presentation\/chen-yuxuan"},{"key":"27_CR8","unstructured":"Core, T.: Simple audio recognition: Recognizing keywords. https:\/\/www.tensorflow.org\/tutorials\/audio\/simple_audio"},{"key":"27_CR9","doi-asserted-by":"publisher","unstructured":"Du, T., Ji, S., Li, J., Gu, Q., Wang, T., Beyah, R.: SirenAttack: generating adversarial audio for end-to-end acoustic systems. In: Proceedings of the 15th ACM Asia Conference on Computer and Communications Security, pp. 357\u2013369. ASIA CCS \u201920, Association for Computing Machinery, New York, NY, USA (2020). https:\/\/doi.org\/10.48550\/arXiv.1901.07846","DOI":"10.48550\/arXiv.1901.07846"},{"key":"27_CR10","doi-asserted-by":"publisher","unstructured":"Eykholt, K., et al.: Robust physical-world attacks on deep learning visual classification. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1625\u20131634 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00175","DOI":"10.1109\/CVPR.2018.00175"},{"key":"27_CR11","unstructured":"Foster, K.: What is Automatic Speech Recognition? A Comprehensive Overview of ASR Technology. https:\/\/www.assemblyai.com\/blog\/what-is-asr"},{"issue":"1","key":"27_CR12","doi-asserted-by":"publisher","first-page":"10713","DOI":"10.1038\/s41598-023-36714-z","volume":"13","author":"T Greer","year":"2023","unstructured":"Greer, T., Shi, X., Ma, B., Narayanan, S.: Creating musical features using multi-faceted, multi-task encoders based on transformers. Sci. Rep. 13(1), 10713 (2023). https:\/\/doi.org\/10.1038\/s41598-023-36714-z","journal-title":"Sci. Rep."},{"key":"27_CR13","doi-asserted-by":"publisher","unstructured":"Guo, H., Wang, G., Wang, Y., Chen, B., Yan, Q., Xiao, L.: Phantomsound: black-box, query-efficient audio adversarial attack via split-second phoneme injection. In: Proceedings of the 26th International Symposium on Research in Attacks, Intrusions and Defenses, pp. 366\u2013380 (2023). https:\/\/doi.org\/10.48550\/arXiv.2309.06960","DOI":"10.48550\/arXiv.2309.06960"},{"key":"27_CR14","unstructured":"Kaggle: Level up with the largest AI & ML community. https:\/\/www.kaggle.com\/"},{"key":"27_CR15","doi-asserted-by":"publisher","unstructured":"Kong, J., Kim, J., Bae, J.: Hifi-gan: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural Inf. Process. Syst. 33, 17022\u201317033 (2020). https:\/\/doi.org\/10.48550\/arXiv.2010.05646","DOI":"10.48550\/arXiv.2010.05646"},{"issue":"1","key":"27_CR16","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1080\/24751839.2022.2132586","volume":"7","author":"F Lepr\u00e9vost","year":"2023","unstructured":"Lepr\u00e9vost, F., Topal, A.O., Avdusinovic, E., Chitic, R.: A strategy creating high-resolution adversarial images against convolutional neural networks and a feasibility study on 10 CNNs. J. Inf. Telecommun. 7(1), 89\u2013119 (2023). https:\/\/doi.org\/10.1080\/24751839.2022.2132586","journal-title":"J. Inf. Telecommun."},{"key":"27_CR17","doi-asserted-by":"publisher","unstructured":"Mogal, F.M.M., Osman\u00a0Topal, A., Mancellari, E., M\u00fcller, V., Avdusinovic, E., Lepr\u00e9vost, F.: Speech-GAN: black-box attack against automatic speech recognition systems. In: 2025 22nd International Joint Conference on Computer Science and Software Engineering (JCSSE), pp. 225\u2013232 (2025). https:\/\/doi.org\/10.1109\/JCSSE67377.2025.11297954","DOI":"10.1109\/JCSSE67377.2025.11297954"},{"key":"27_CR18","doi-asserted-by":"publisher","unstructured":"Nercessian, S., Sarroff, A., Werner, K.J.: Lightweight and interpretable neural modeling of an audio distortion effect using hyperconditioned differentiable biquads. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 890\u2013894. IEEE (2021).https:\/\/doi.org\/10.48550\/arXiv.2103.08709","DOI":"10.48550\/arXiv.2103.08709"},{"key":"27_CR19","doi-asserted-by":"publisher","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: An ASR corpus based on public domain audiobooks. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5206\u20135210. IEEE (2015). https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"27_CR20","doi-asserted-by":"publisher","unstructured":"Puvvada, K.C., Koluguri, N.R., Dhawan, K., Balam, J., Ginsburg, B.: Discrete audio representation as an alternative to mel-spectrograms for speaker and speech recognition. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 12111\u201312115. IEEE (2024). https:\/\/doi.org\/10.48550\/arXiv.2309.10922","DOI":"10.48550\/arXiv.2309.10922"},{"key":"27_CR21","unstructured":"PyTorch: TORCHAUDIO.FUNCTIONAL.BANDPASS_BIQUAD. https:\/\/docs.pytorch.org\/audio\/main\/generated\/torchaudio.functional.bandpass_biquad.html"},{"key":"27_CR22","doi-asserted-by":"publisher","unstructured":"Qin, Y., Carlini, N., Cottrell, G., Goodfellow, I., Raffel, C.: Imperceptible, robust, and targeted adversarial examples for automatic speech recognition. In: International conference on machine learning, pp. 5231\u20135240. PMLR (2019). https:\/\/doi.org\/10.48550\/arXiv.1903.10346","DOI":"10.48550\/arXiv.1903.10346"},{"key":"27_CR23","doi-asserted-by":"publisher","unstructured":"Sharif, M., Bhagavatula, S., Bauer, L., Reiter, M.K.: Accessorize to a crime: real and stealthy attacks on state-of-the-art face recognition. In: Proceedings of the 2016 ACM SIGSAC Conference on Computer and Communications Security, pp. 1528\u20131540 (2016). https:\/\/doi.org\/10.1145\/2976749.2978392","DOI":"10.1145\/2976749.2978392"},{"key":"27_CR24","doi-asserted-by":"publisher","unstructured":"Taori, R., Kamsetty, A., Chu, B., Vemuri, N.: Targeted adversarial examples for black box audio systems. In: 2019 IEEE security and privacy workshops (SPW), pp. 15\u201320. IEEE (2019). https:\/\/doi.org\/10.48550\/arXiv.1805.07820","DOI":"10.48550\/arXiv.1805.07820"},{"issue":"8","key":"27_CR25","doi-asserted-by":"publisher","first-page":"3493","DOI":"10.3390\/app14083493","volume":"14","author":"AO Topal","year":"2024","unstructured":"Topal, A.O., Mancellari, E., Lepr\u00e9vost, F., Avdusinovic, E., Gillet, T.: The noise blowing-up strategy creates high quality high resolution adversarial images against convolutional neural networks. Appl. Sci. 14(8), 3493 (2024). https:\/\/doi.org\/10.3390\/app14083493","journal-title":"Appl. Sci."},{"key":"27_CR26","unstructured":"Toy, R.: Audio EQ Cookbook (2021). https:\/\/www.w3.org\/TR\/audio-eq-cookbook\/"},{"key":"27_CR27","doi-asserted-by":"publisher","unstructured":"Varrette, S., Bouvry, P., Cartiaux, H., Georgatos, F.: Management of an academic HPC cluster: the UL experience. In: Proceedings of the 2014 Intl. Conf. on High Performance Computing & Simulation (HPCS 2014), pp. 959\u2013967. IEEE, Bologna, Italy (2014). https:\/\/doi.org\/10.1109\/HPCSim.2014.6903792","DOI":"10.1109\/HPCSim.2014.6903792"},{"key":"27_CR28","doi-asserted-by":"publisher","unstructured":"Wang, Q., Zheng, B., Li, Q., Shen, C., Ba, Z.: Towards query-efficient adversarial attacks against automatic speech recognition systems. IEEE Trans. Inf. Forens. Sec. 16, 896\u2013908 (2020). https:\/\/doi.org\/10.1109\/TIFS.2020.3026543","DOI":"10.1109\/TIFS.2020.3026543"},{"key":"27_CR29","doi-asserted-by":"publisher","unstructured":"Yamamoto, R., Song, E., Kim, J.M.: Parallel WaveGAN: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6199\u20136203. IEEE (2020). https:\/\/doi.org\/10.48550\/arXiv.1910.11480","DOI":"10.48550\/arXiv.1910.11480"},{"key":"27_CR30","doi-asserted-by":"publisher","unstructured":"Yuan, X., et al.: CommanderSong: a systematic approach for practical adversarial voice recognition. In: 27th USENIX Security Symposium (USENIX Security 18), pp. 49\u201364 (2018). https:\/\/doi.org\/10.48550\/arXiv.1801.08535","DOI":"10.48550\/arXiv.1801.08535"},{"key":"27_CR31","doi-asserted-by":"publisher","unstructured":"Zheng, B., et al.: Black-box Adversarial Attacks on Commercial Speech Platforms with Minimal Information. In: Proceedings of the 2021 ACM SIGSAC Conference on Computer and Communications Security, pp. 86\u2013107 (2021). https:\/\/doi.org\/10.48550\/arXiv.2110.09714","DOI":"10.48550\/arXiv.2110.09714"}],"container-title":["Lecture Notes in Computer Science","Intelligent Information and Database Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-0074-0_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T10:52:56Z","timestamp":1780397576000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-0074-0_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819200733","9789819200740"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-0074-0_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ACIIDS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Intelligent Information and Database Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kaohsiung","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiwan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 April 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aciids2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/aciids.pwr.edu.pl\/2026\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}