{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T12:57:58Z","timestamp":1781009878623,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,2,6]],"date-time":"2024-02-06T00:00:00Z","timestamp":1707177600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"EU Horizon 2020","award":["101007350"],"award-info":[{"award-number":["101007350"]}]},{"name":"State Government of Styria, Austria ? Department Zukunftsfonds Steiermark"},{"name":"Silicon Austria Labs (SAL) - University SAL Lab initiative"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,2,6]]},"DOI":"10.1145\/3597503.3623311","type":"proceedings-article","created":{"date-parts":[[2024,2,6]],"date-time":"2024-02-06T20:53:16Z","timestamp":1707252796000},"page":"1-13","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Learning and Repair of Deep Reinforcement Learning Policies from Fuzz-Testing Data"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4193-5609","authenticated-orcid":false,"given":"Martin","family":"Tappler","sequence":"first","affiliation":[{"name":"Institute of Software Technology, Graz University of Technology, Graz, Austria"},{"name":"TU Graz - SAL, DES Lab, Silicon Austria Labs, Graz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9446-9541","authenticated-orcid":false,"given":"Andrea","family":"Pferscher","sequence":"additional","affiliation":[{"name":"Institute of Software Technology, Graz University of Technology, Graz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3484-5584","authenticated-orcid":false,"given":"Bernhard K.","family":"Aichernig","sequence":"additional","affiliation":[{"name":"Institute of Software Technology, Graz University of Technology, Graz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5183-5452","authenticated-orcid":false,"given":"Bettina","family":"K\u00f6nighofer","sequence":"additional","affiliation":[{"name":"Institute of Applied Information Processing and Communications, Graz University of Technology, Graz, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,2,6]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-31280-0_1"},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems 31: Annual Conference on Neural Information Processing Systems 2018","author":"Aytar Yusuf","year":"2018","unstructured":"Yusuf Aytar, Tobias Pfaff, David Budden, Tom Le Paine, Ziyu Wang, and Nando de Freitas. 2018. Playing Hard Exploration Games by Watching YouTube. In Advances in Neural Information Processing Systems 31: Annual Conference on Neural Information Processing Systems 2018, NeurIPS 2018, December 3--8, 2018, Montr\u00e9al, Canada. 2935--2945. https:\/\/proceedings.neurips.cc\/paper\/2018\/hash\/35309226eb45ec366ca86a4329a2b7c3-Abstract.html"},{"key":"e_1_3_2_1_3_1","unstructured":"Matteo Biagiola and Paolo Tonella. 2023. Testing of Deep Reinforcement Learning Agents with Surrogate Models. arXiv:2305.12751 [cs.SE]"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-019-05849-4"},{"key":"e_1_3_2_1_5_1","volume-title":"Pablo Samuel Castro, and Jordan Terry","author":"Chevalier-Boisvert Maxime","year":"2023","unstructured":"Maxime Chevalier-Boisvert, Bolun Dai, Mark Towers, Rodrigo de Lazcano, Lucas Willems, Salem Lahlou, Suman Pal, Pablo Samuel Castro, and Jordan Terry. 2023. Minigrid & Miniworld: Modular & Customizable Reinforcement Learning Environments for Goal-Oriented Tasks. CoRR abs\/2306.13831 (2023)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13748-019-00203-0"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3324884.3416592"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.12372"},{"key":"e_1_3_2_1_9_1","volume-title":"6th International Conference on Learning Representations, ICLR","author":"Gao Yang","year":"2018","unstructured":"Yang Gao, Huazhe Xu, Ji Lin, Fisher Yu, Sergey Levine, and Trevor Darrell. 2018. Reinforcement Learning from Imperfect Demonstrations. In 6th International Conference on Learning Representations, ICLR 2018, Vancouver, BC, Canada, April 30 -- May 3, 2018, Workshop Track Proceedings. OpenReview.net. https:\/\/openreview.net\/forum?id=HytbCQG8z"},{"key":"e_1_3_2_1_10_1","volume-title":"Brafman","author":"Gaon Maor","year":"2020","unstructured":"Maor Gaon and Ronen I. Brafman. 2020. Reinforcement Learning with Non-Markovian Rewards. In The Thirty-Fourth AAAI Conference on Artificial Intelligence, AAAI 2020, The Thirty-Second Innovative Applications of Artificial Intelligence Conference, IAAI 2020, The Tenth AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI 2020, New York, NY, USA, February 7--12, 2020. AAAI Press, 3980--3987. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/5814"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.5555\/3398761.3398819"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/MS.2021.3072577"},{"key":"e_1_3_2_1_13_1","volume-title":"Alessandro Abate, Tom Melham, and Daniel Kroening.","author":"Hasanbeig Mohammadhosein","year":"2021","unstructured":"Mohammadhosein Hasanbeig, Natasha Yogananda Jeppu, Alessandro Abate, Tom Melham, and Daniel Kroening. 2021. DeepSynth: Automata Synthesis for Automatic Task Segmentation in Deep Reinforcement Learning. In Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Thirty-Third Conference on Innovative Applications of Artificial Intelligence, IAAI 2021, The Eleventh Symposium on Educational Advances in Artificial Intelligence, EAAI 2021, Virtual Event, February 2--9, 2021. AAAI Press, 7647--7656. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/16935"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11757"},{"key":"e_1_3_2_1_15_1","volume-title":"Learning Reward Machines for Partially Observable Reinforcement Learning. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019","author":"Icarte Rodrigo Toro","year":"2019","unstructured":"Rodrigo Toro Icarte, Ethan Waldie, Toryn Q. Klassen, Richard Anthony Valenzano, Margarita P. Castro, and Sheila A. McIlraith. 2019. Learning Reward Machines for Partially Observable Reinforcement Learning. In Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, NeurIPS 2019, December 8--14, 2019, Vancouver, BC, Canada, Hanna M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d'Alch\u00e9-Buc, Emily B. Fox, and Roman Garnett (Eds.). 15497--15508. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/532435c44bec236b471a47a88d63513d-Abstract.html"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5953"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning, ICML 2018","author":"Kang Bingyi","year":"2018","unstructured":"Bingyi Kang, Zequn Jie, and Jiashi Feng. 2018. Policy Optimization with Demonstrations. In Proceedings of the 35th International Conference on Machine Learning, ICML 2018, Stockholmsmassan, Stockholm, Sweden, July 10--15, 2018 (Proceedings of Machine Learning Research, Vol. 80). PMLR, 2474--2483. http:\/\/proceedings.mlr.press\/v80\/kang18a.html"},{"key":"e_1_3_2_1_18_1","unstructured":"Christian Kauten. 2018. Super Mario Bros for OpenAI Gym. GitHub. https:\/\/github.com\/Kautenja\/gym-super-mario-bros"},{"key":"e_1_3_2_1_19_1","volume-title":"Learning Nondeterministic Mealy Machines. In ICGI 2014 (JMLR Workshop and Conference Proceedings","volume":"123","author":"Khalili Ali","year":"2014","unstructured":"Ali Khalili and Armando Tacchella. 2014. Learning Nondeterministic Mealy Machines. In ICGI 2014 (JMLR Workshop and Conference Proceedings, Vol. 34). 109--123. http:\/\/proceedings.mlr.press\/v34\/khalili14a.html"},{"key":"e_1_3_2_1_20_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7--9, 2015, Conference Track Proceedings. http:\/\/arxiv.org\/abs\/1412.6980"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2773081"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysarc.2022.102701"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-96562-8_2"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20753"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE.2013.6606623"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"4911","author":"Odena Augustus","year":"2019","unstructured":"Augustus Odena, Catherine Olsson, David Andersen, and Ian Goodfellow. 2019. TensorFuzz: Debugging Neural Networks with Coverage-Guided Fuzzing. In Proceedings of the 36th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 97), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.). PMLR, 4901--4911."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1018076709321"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-44851-9_35"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-10085-1"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2970619"},{"key":"e_1_3_2_1_33_1","volume-title":"Courville","author":"Schwarzer Max","year":"2021","unstructured":"Max Schwarzer, Nitarshan Rajkumar, Michael Noukhovitch, Ankesh Anand, Laurent Charlin, R. Devon Hjelm, Philip Bachman, and Aaron C. Courville. 2021. Pretraining Representations for Data-Efficient Reinforcement Learning. CoRR abs\/2106.04799 (2021). arXiv:2106.04799 https:\/\/arxiv.org\/abs\/2106.04799"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"e_1_3_2_1_35_1","volume-title":"Barto","author":"Sutton Richard S.","year":"1998","unstructured":"Richard S. Sutton and Andrew G. Barto. 1998. Reinforcement Learning - An Introduction. MIT Press. https:\/\/www.worldcat.org\/oclc\/37293240"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/72"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.6084\/m9.figshare.22353712"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273607"},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9--15","volume":"6224","author":"Tessler Chen","year":"2019","unstructured":"Chen Tessler, Yonathan Efroni, and Shie Mannor. 2019. Action Robust Reinforcement Learning and Applications in Continuous Control. In Proceedings of the 36th International Conference on Machine Learning, ICML 2019, 9--15 June 2019, Long Beach, California, USA (Proceedings of Machine Learning Research, Vol. 97), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.). PMLR, 6215--6224. http:\/\/proceedings.mlr.press\/v97\/tessler19a.html"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3180155.3180220"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"e_1_3_2_1_42_1","volume-title":"Max Tegmark, and Francesco Fuso Nerini.","author":"Vinuesa Ricardo","year":"2020","unstructured":"Ricardo Vinuesa, Hossein Azizpour, Iolanda Leite, Madeline Balaam, Virginia Dignum, Sami Domisch, Anna Fell\u00e4nder, Simone Daniela Langhans, Max Tegmark, and Francesco Fuso Nerini. 2020. The Role of Artificial Intelligence in Achieving the Sustainable Development Goals. Nature communications 11, 1 (2020), 233."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","unstructured":"Oriol Vinyals Igor Babuschkin Wojciech M. Czarnecki Micha\u00ebl Mathieu Andrew Dudzik Junyoung Chung David H. Choi Richard Powell Timo Ewalds Petko Georgiev Junhyuk Oh Dan Horgan Manuel Kroiss Ivo Danihelka Aja Huang Laurent Sifre Trevor Cai John P. Agapiou Max Jaderberg Alexander Sasha Vezhnevets R\u00e9mi Leblond Tobias Pohlen Valentin Dalibard David Budden Yury Sulsky James Molloy Tom Le Paine \u00c7aglar G\u00fcl\u00e7ehre Ziyu Wang Tobias Pfaff Yuhuai Wu Roman Ring Dani Yogatama Dario W\u00fcnsch Katrina McKinney Oliver Smith Tom Schaul Timothy P. Lillicrap Koray Kavukcuoglu Demis Hassabis Chris Apps and David Silver. 2019. Grandmaster Level in StarCraft II using Multi-Agent Reinforcement Learning. Nat. 575 7782 (2019) 350--354. 10.1038\/s41586-019-1724-z","DOI":"10.1038\/s41586-019-1724-z"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3293882.3330579"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1609\/icaps.v30i1.6756"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/820"},{"key":"e_1_3_2_1_47_1","volume-title":"The Fuzzing Book. https:\/\/www.fuzzingbook.org\/. accessed","author":"Zeller Andreas","year":"2023","unstructured":"Andreas Zeller, Rahul Gopinath, Marcel B\u00f6hme, Gordon Fraser, and Christian Holler. 2021. The Fuzzing Book. https:\/\/www.fuzzingbook.org\/. accessed: 2023, August 31."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20914"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","unstructured":"Amirhossein Zolfagharian Manel Abdellatif Lionel C. Briand Mojtaba Bagherzadeh and Ramesh S. 2022. Search-Based Testing Approach for Deep Reinforcement Learning Agents. CoRR abs\/2206.07813 (2022). arXiv:2206.07813 10.48550\/arXiv.2206.07813","DOI":"10.48550\/arXiv.2206.07813"}],"event":{"name":"ICSE '24: IEEE\/ACM 46th International Conference on Software Engineering","location":"Lisbon Portugal","acronym":"ICSE '24","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS","Faculty of Engineering of University of Porto"]},"container-title":["Proceedings of the IEEE\/ACM 46th International Conference on Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3597503.3623311","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3597503.3623311","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:48:45Z","timestamp":1750182525000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3597503.3623311"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,6]]},"references-count":49,"alternative-id":["10.1145\/3597503.3623311","10.1145\/3597503"],"URL":"https:\/\/doi.org\/10.1145\/3597503.3623311","relation":{},"subject":[],"published":{"date-parts":[[2024,2,6]]},"assertion":[{"value":"2024-02-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}