{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:05:04Z","timestamp":1782864304161,"version":"3.54.5"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032161642","type":"print"},{"value":"9783032161659","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-16165-9_2","type":"book-chapter","created":{"date-parts":[[2026,4,19]],"date-time":"2026-04-19T22:56:29Z","timestamp":1776639389000},"page":"19-38","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Adversarial Evasion Against Autonomous Cyber Defence Agents"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-9716-8658","authenticated-orcid":false,"given":"Melanie","family":"Meijer","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sanyam","family":"Vyas","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vasilios","family":"Mavroudis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marc","family":"Juarez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,1]]},"reference":[{"key":"2_CR1","doi-asserted-by":"publisher","first-page":"50377","DOI":"10.52202\/075280-2192","volume":"36","author":"D Abel","year":"2023","unstructured":"Abel, D., Barreto, A., Van Roy, B., Precup, D., van Hasselt, H.P., Singh, S.: A definition of continual reinforcement learning. Adv. Neural. Inf. Process. Syst. 36, 50377\u201350407 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2_CR2","unstructured":"Andrew, A., Spillard, S., Collyer, J., Dhir, N.: Developing optimal causal cyber-defence agents via cyber security simulation. arXiv preprint arXiv:2207.12355 (2022)"},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Bai, F., Liu, R., Du, Y., Wen, Y., Yang, Y.: Rat: Adversarial attacks on deep reinforcement agents for targeted behaviors. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a039, pp. 15453\u201315461 (2025)","DOI":"10.1609\/aaai.v39i15.33696"},{"key":"2_CR4","unstructured":"Bates, E., Hicks, C., Mavroudis, V.: Less is more? rewards in RL for cyber defence. arXiv preprint arXiv:2503.03245 (2025)"},{"issue":"6","key":"2_CR5","doi-asserted-by":"publisher","first-page":"503","DOI":"10.1090\/S0002-9904-1954-09848-8","volume":"60","author":"R Bellman","year":"1954","unstructured":"Bellman, R.: The theory of dynamic programming. Bull. Am. Math. Soc. 60(6), 503\u2013515 (1954)","journal-title":"Bull. Am. Math. Soc."},{"key":"2_CR6","unstructured":"Bengio, Y., et\u00a0al.: International scientific report on the safety of advanced AI (interim report). arXiv preprint arXiv:2412.05282 (2024)"},{"key":"2_CR7","unstructured":"Berner, C., et\u00a0al.: Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680 (2019)"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Collins, J., Howard, D., Leitner, J.: Quantifying the reality gap in robotic manipulation tasks. In: 2019 International Conference on Robotics and Automation (ICRA). pp. 6706\u20136712. IEEE (2019)","DOI":"10.1109\/ICRA.2019.8793591"},{"key":"2_CR9","unstructured":"Das, A., Rad, P.: Opportunities and challenges in explainable artificial intelligence (XAI): A survey. arXiv preprint arXiv:2006.11371 (2020)"},{"key":"2_CR10","doi-asserted-by":"crossref","unstructured":"Ell, M., Rizvi, S.: Cyber security breaches survey 2024 (2024). https:\/\/www.gov.uk\/government\/statistics\/cyber-security-breaches-survey-2024\/cyber-security-breaches-survey-2024#summary","DOI":"10.63180\/jcsra.thestap.2024.1.3"},{"key":"2_CR11","unstructured":"Emerson, H., Bates, L., Hicks, C., Mavroudis, V.: Cyborg++: An enhanced gym for the development of autonomous cyber agents. arXiv preprint arXiv:2410.16324 (2024)"},{"key":"2_CR12","doi-asserted-by":"publisher","unstructured":"Foley, M., Hicks, C., Highnam, K., Mavroudis, V.: Autonomous network defence using reinforcement learning. In: Proceedings of the 2022 ACM on Asia Conference on Computer and Communications Security. p. 1252\u20131254. ASIA CCS \u201922, Association for Computing Machinery, New York, NY, USA (2022). https:\/\/doi.org\/10.1145\/3488932.3527286, https:\/\/doi.org\/10.1145\/3488932.3527286","DOI":"10.1145\/3488932.3527286"},{"key":"2_CR13","unstructured":"Gleave, A., Dennis, M., Wild, C., Kant, N., Levine, S., Russell, S.: Adversarial policies: Attacking deep reinforcement learning. arXiv preprint arXiv:1905.10615 (2019)"},{"key":"2_CR14","unstructured":"Goodfellow, I.J., Shlens, J., Szegedy, C.: Explaining and harnessing adversarial examples. arXiv preprint arXiv:1412.6572 (2014)"},{"key":"2_CR15","unstructured":"Guo, W., Wu, X., Wang, L., Xing, X., Song, D.: $$\\{$$PATROL$$\\}$$: Provable defense against adversarial policy in two-player games. In: 32nd USENIX Security Symposium (USENIX Security 23). pp. 3943\u20133960 (2023)"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Hu, Z., Beuran, R., Tan, Y.: Automated penetration testing using deep reinforcement learning. In: 2020 IEEE European Symposium on Security and Privacy Workshops (EuroS&PW). pp. 2\u201310. IEEE (2020)","DOI":"10.1109\/EuroSPW51379.2020.00010"},{"key":"2_CR17","unstructured":"Kiely, M., Bowman, D., Standen, M., Moir, C.: On autonomous agents in a cyber defence environment. arXiv preprint arXiv:2309.07388 (2023)"},{"key":"2_CR18","doi-asserted-by":"crossref","unstructured":"Lee, X.Y., Ghadai, S., Tan, K.L., Hegde, C., Sarkar, S.: Spatiotemporally constrained action space attacks on deep reinforcement learning agents. In: Proceedings of the AAAI conference on artificial intelligence. vol.\u00a034, pp. 4577\u20134584 (2020)","DOI":"10.1609\/aaai.v34i04.5887"},{"key":"2_CR19","unstructured":"Liu, X., Chakraborty, S., Sun, Y., Huang, F.: Rethinking adversarial policies: A generalized attack formulation and provable defense in RL. arXiv preprint arXiv:2305.17342 (2023)"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Loevenich, J., Adler, E., Huerten, T., Lopes, R.R.F.: Design and evaluation of an autonomous cyber defence agent using DRL and an augmented llm. Comput. Netw. 262, 111162 (2025)","DOI":"10.1016\/j.comnet.2025.111162"},{"key":"2_CR21","unstructured":"Madry, A., Makelov, A., Schmidt, L., Tsipras, D., Vladu, A.: Towards deep learning models resistant to adversarial attacks. arXiv preprint arXiv:1706.06083 (2017)"},{"key":"2_CR22","unstructured":"Meijer, M.: Automated Cyber Defence by Deep Reinforcement Learning. Bachelor\u2019s thesis, Cardiff University (2023)"},{"key":"2_CR23","unstructured":"Molina-Markham, A., Miniter, C., Powell, B., Ridley, A.: Network environment design for autonomous cyberdefense. arXiv preprint arXiv:2103.07583 (2021)"},{"key":"2_CR24","unstructured":"Narvekar, S., Peng, B., Leonetti, M., Sinapov, J., Taylor, M.E., Stone, P.: Curriculum learning for reinforcement learning domains: a framework and survey. J. Mach. Learn. Res. 21(181), 1\u201350 (2020)"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Potteiger, N., Samaddar, A., Bergstrom, H., Koutsoukos, X.: Designing robust cyber-defense agents with evolving behavior trees. In: 2024 International Conference on Assured Autonomy (ICAA). pp. 1\u201310. IEEE (2024)","DOI":"10.1109\/ICAA64256.2024.00011"},{"key":"2_CR26","unstructured":"Schott, L., et al.: Robust deep reinforcement learning through adversarial attacks and training: a survey. arXiv preprint arXiv:2403.00420 (2024)"},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Schott, L., Hajri, H., Lamprier, S.: Improving robustness of deep reinforcement learning agents: environment attack based on the critic network. In: 2022 International Joint Conference on Neural Networks (IJCNN). pp.\u00a01\u20138. IEEE (2022)","DOI":"10.1109\/IJCNN55064.2022.9892901"},{"key":"2_CR28","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"2_CR29","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"2_CR30","unstructured":"Schwartz, J.: Autonomous penetration testing using reinforcement learning (2018). URL for simulator documentation: https:\/\/networkattacksimulator.readthedocs.io\/en\/latest\/"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Silver, D., et al.: A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419), 1140\u20131144 (2018)","DOI":"10.1126\/science.aar6404"},{"key":"2_CR32","unstructured":"Standen, M., Lucas, M., Bowman, D., Richer, T.J., Kim, J., Marriott, D.: Cyborg: A gym for the development of autonomous cyber agents. arXiv preprint arXiv:2108.09118 (2021)"},{"key":"2_CR33","unstructured":"Standen, M., Lucas, M., Bowman, D., Richer, T.J., Kim, J., Marriott, D.: Cyborg: A gym for the development of autonomous cyber agents. arXiv preprint arXiv:2108.09118 (2021)"},{"key":"2_CR34","unstructured":"Szegedy, C., et al.: Intriguing properties of neural networks. arXiv preprint arXiv:1312.6199 (2013)"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Vyas, S., Mavroudis, V., Burnap, P.: Towards the deployment of realistic autonomous cyber network defence: a systematic review. ACM Comput. Sur. (2025)","DOI":"10.1145\/3729213"},{"key":"2_CR36","unstructured":"Wang, T.T., et\u00a0al.: Adversarial policies beat superhuman go ais. In: International Conference on Machine Learning. pp. 35655\u201335739. PMLR (2023)"},{"issue":"4","key":"2_CR37","doi-asserted-by":"publisher","first-page":"2751","DOI":"10.1109\/JIOT.2019.2957289","volume":"7","author":"L Yu","year":"2019","unstructured":"Yu, L., et al.: Deep reinforcement learning for smart home energy management. IEEE Internet Things J. 7(4), 2751\u20132762 (2019)","journal-title":"IEEE Internet Things J."},{"key":"2_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.cose.2023.103578","volume":"136","author":"Z Zhu","year":"2024","unstructured":"Zhu, Z., Chen, M., Zhu, C., Zhu, Y.: Effective defense strategies in network security using improved double dueling deep q-network. Comput. & Sec. 136, 103578 (2024)","journal-title":"Comput. & Sec."}],"container-title":["Lecture Notes in Computer Science","Computer Security. ESORICS 2025 International Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-16165-9_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T23:39:01Z","timestamp":1782862741000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-16165-9_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032161642","9783032161659"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-16165-9_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 April 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ESORICS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Symposium on Research in Computer Security","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toulouse","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"esorics2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.esorics2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}