{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T13:52:55Z","timestamp":1787061175672,"version":"3.56.0"},"publisher-location":"Cham","reference-count":43,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031757778","type":"print"},{"value":"9783031757785","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,18]],"date-time":"2024-11-18T00:00:00Z","timestamp":1731888000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,18]],"date-time":"2024-11-18T00:00:00Z","timestamp":1731888000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-75778-5_2","type":"book-chapter","created":{"date-parts":[[2024,11,17]],"date-time":"2024-11-17T07:09:24Z","timestamp":1731827364000},"page":"18-38","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Safe Reinforcement Learning Through Regret and\u00a0State Restorations in\u00a0Evaluation Stages"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1100-1952","authenticated-orcid":false,"given":"Timo P.","family":"Gros","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5932-3395","authenticated-orcid":false,"given":"Nicola J.","family":"M\u00fcller","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2776-9288","authenticated-orcid":false,"given":"Daniel","family":"H\u00f6ller","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8460-6007","authenticated-orcid":false,"given":"Verena","family":"Wolf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,18]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Alshiekh, M., Bloem, R., Ehlers, R., K\u00f6nighofer, B., Niekum, S., Topcu, U.: Safe reinforcement learning via shielding. In: Proceedings of the 32nd AAAI Conference on Artificial Intelligence (AAAI), pp. 2669\u20132678. AAAI Press (2018)","DOI":"10.1609\/aaai.v32i1.11797"},{"key":"2_CR2","unstructured":"Amit, R., Meir, R., Ciosek, K.: Discount factor as a regularizer in reinforcement learning. In: Proceedings of the 37th International Conference on Machine Learning (ICML), pp. 269\u2013278. PMLR (2020)"},{"key":"2_CR3","unstructured":"Anderson, G., Chaudhuri, S., Dillig, I.: Guiding safe exploration with weakest preconditions. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"2_CR4","unstructured":"Andrychowicz, M., et al.: Hindsight experience replay. In: Advances in Neural Information Processing Systems, pp. 5048\u20135058 (2017)"},{"key":"2_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"630","DOI":"10.1007\/978-3-030-25540-4_36","volume-title":"Computer Aided Verification","author":"G Avni","year":"2019","unstructured":"Avni, G., Bloem, R., Chatterjee, K., Henzinger, T.A., K\u00f6nighofer, B., Pranger, S.: Run-time optimization for learned controllers through quantitative games. In: Dillig, I., Tasiran, S. (eds.) CAV 2019. LNCS, vol. 11561, pp. 630\u2013649. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-25540-4_36"},{"key":"2_CR6","unstructured":"Azar, M.G., Osband, I., Munos, R.: Minimax regret bounds for reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning (ICML), pp. 263\u2013272. PMLR (2017)"},{"key":"2_CR7","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/978-3-030-73959-1_8","volume-title":"Trustworthy AI - Integrating Learning, Optimization and Reasoning","author":"C Baier","year":"2021","unstructured":"Baier, C., Christakis, M., Gros, T.P., Gro\u00df, D., Gumhold, S., Hermanns, H., Hoffmann, J., Klauck, M.: Lab conditions for research on\u00a0explainable automated decisions. In: Heintz, F., Milano, M., O\u2019Sullivan, B. (eds.) TAILOR 2020. LNCS (LNAI), vol. 12641, pp. 83\u201390. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-73959-1_8"},{"key":"2_CR8","unstructured":"Bharadhwaj, H., Kumar, A., Rhinehart, N., Levine, S., Shkurti, F., Garg, A.: Conservative safety critics for exploration. In: Proceedings of the 9th International Conference on Learning Representations (ICLR). OpenReview (2021)"},{"key":"2_CR9","unstructured":"Burda, Y., Edwards, H., Storkey, A.J., Klimov, O.: Exploration by random network distillation. In: Proceedings of the 7th International Conference on Learning Representations (ICLR). OpenReview (2019)"},{"key":"2_CR10","unstructured":"Campero, A., Raileanu, R., K\u00fcttler, H., Tenenbaum, J.B., Rockt\u00e4schel, T., Grefenstette, E.: Learning with AMIGo: adversarially motivated intrinsic goals. In: Proceedings of the 9th International Conference on Learning Representations (ICLR). OpenReview (2021)"},{"key":"2_CR11","unstructured":"Chevalier-Boisvert, M., et al.: BabyAI: a platform to study the sample efficiency of grounded language learning. In: Proceedings of the 7th International Conference on Learning Representations (ICLR). OpenReview (2019)"},{"issue":"7847","key":"2_CR12","doi-asserted-by":"publisher","first-page":"580","DOI":"10.1038\/s41586-020-03157-9","volume":"590","author":"A Ecoffet","year":"2021","unstructured":"Ecoffet, A., Huizinga, J., Lehman, J., Stanley, K.O., Clune, J.: First return, then explore. Nature 590(7847), 580\u2013586 (2021)","journal-title":"Nature"},{"key":"2_CR13","unstructured":"Flet-Berliac, Y., Ferret, J., Pietquin, O., Preux, P., Geist, M.: Adversarially guided actor-critic. In: Proceedings of the 9th International Conference on Learning Representations (ICLR). OpenReview (2021)"},{"key":"2_CR14","unstructured":"Fujita, Y., Nagarajan, P., Kataoka, T., Ishikawa, T.: ChainerRL: a deep reinforcement learning library. J. Mach. Learn. Res. 22, 77:1\u201377:14 (2021)"},{"key":"2_CR15","first-page":"1437","volume":"16","author":"J Garc\u00eda","year":"2015","unstructured":"Garc\u00eda, J., Fern\u00e1ndez, F.: A comprehensive survey on safe reinforcement learning. J. Mach. Learn. Res. 16, 1437\u20131480 (2015)","journal-title":"J. Mach. Learn. Res."},{"key":"2_CR16","doi-asserted-by":"publisher","unstructured":"Gros, T.P., et al.: DSMC evaluation stages: fostering robust and safe behavior in deep reinforcement learning - extended version. ACM Trans. Model. Comput. Simulat. 33(4), 17:1\u201317:28 (2023). https:\/\/doi.org\/10.1145\/3607198","DOI":"10.1145\/3607198"},{"key":"2_CR17","doi-asserted-by":"publisher","unstructured":"Gros, T.P., Hermanns, H., Hoffmann, J., Klauck, M., K\u00f6hl, M.A., Wolf, V.: MoGym: using formal models for training and verifying decision-making agents. In: Shoham, S., Vizel, Y. (eds.) CAV 2022, Part II, pp. 430\u2013443. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-13188-2_21","DOI":"10.1007\/978-3-031-13188-2_21"},{"key":"2_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"96","DOI":"10.1007\/978-3-030-50086-3_6","volume-title":"Formal Techniques for Distributed Objects, Components, and Systems","author":"TP Gros","year":"2020","unstructured":"Gros, T.P., Hermanns, H., Hoffmann, J., Klauck, M., Steinmetz, M.: Deep statistical model checking. In: Gotsman, A., Sokolova, A. (eds.) FORTE 2020. LNCS, vol. 12136, pp. 96\u2013114. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-50086-3_6"},{"issue":"3","key":"2_CR19","doi-asserted-by":"publisher","first-page":"407","DOI":"10.1007\/s10009-022-00685-9","volume":"25","author":"TP Gros","year":"2023","unstructured":"Gros, T.P., Hermanns, H., Hoffmann, J., Klauck, M., Steinmetz, M.: Analyzing neural network behavior through deep statistical model checking. Int. J. Softw. Tools Technol. Transfer 25(3), 407\u2013426 (2023)","journal-title":"Int. J. Softw. Tools Technol. Transfer"},{"key":"2_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1007\/978-3-030-85172-9_11","volume-title":"Quantitative Evaluation of Systems","author":"TP Gros","year":"2021","unstructured":"Gros, T.P., H\u00f6ller, D., Hoffmann, J., Klauck, M., Meerkamp, H., Wolf, V.: DSMC evaluation stages: fostering robust and safe behavior in deep reinforcement learning. In: Abate, A., Marin, A. (eds.) QEST 2021. LNCS, vol. 12846, pp. 197\u2013216. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-85172-9_11"},{"key":"2_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/978-3-030-59854-9_2","volume-title":"Quantitative Evaluation of Systems","author":"TP Gros","year":"2020","unstructured":"Gros, T.P., H\u00f6ller, D., Hoffmann, J., Wolf, V.: Tracking the race between deep reinforcement learning and imitation learning. In: Gribaudo, M., Jansen, D.N., Remke, A. (eds.) QEST 2020. LNCS, vol. 12289, pp. 11\u201317. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59854-9_2"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Gu, S., Holly, E., Lillicrap, T.P., Levine, S.: Deep reinforcement learning for robotic manipulation with asynchronous off-policy updates. In: Proceedings of the IEEE International Conference on Robotics and Automation (ICRA), pp. 3389\u20133396. IEEE Press (2017)","DOI":"10.1109\/ICRA.2017.7989385"},{"key":"2_CR23","unstructured":"Hare, J.: Dealing with sparse rewards in reinforcement learning. arXiv preprint arXiv:1910.09281 (2019)"},{"key":"2_CR24","unstructured":"Hasanbeig, M., Abate, A., Kroening, D.: Logically-constrained reinforcement learning. arXiv preprint arXiv:1801.08099 (2018)"},{"key":"2_CR25","unstructured":"Jansen, N., K\u00f6nighofer, B., Junges, S., Serban, A., Bloem, R.: Safe reinforcement learning using probabilistic shields. In: Proceedings of the 31st International Conference on Concurrency Theory (CONCUR), pp. 3:1\u20133:16. Schloss Dagstuhl \u2013 Leibniz-Zentrum f\u00fcr Informatik (2020)"},{"key":"2_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"576","DOI":"10.1007\/978-3-642-39799-8_38","volume-title":"Computer Aided Verification","author":"C Jegourel","year":"2013","unstructured":"Jegourel, C., Legay, A., Sedwards, S.: Importance splitting for statistical model checking rare properties. In: Sharygina, N., Veith, H. (eds.) CAV 2013. LNCS, vol. 8044, pp. 576\u2013591. Springer, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-39799-8_38"},{"key":"2_CR27","unstructured":"Jiang, M., Dennis, M., Parker-Holder, J., Foerster, J.N., Grefenstette, E., Rockt\u00e4schel, T.: Replay-guided adversarial environment design. In: Proceedings of the Annual Conference on Neural Information Processing Systems (NeurIPS), pp. 1884\u20131897 (2021)"},{"key":"2_CR28","unstructured":"Kirkpatrick, J., et al.: Overcoming catastrophic forgetting in neural networks. arXiv preprint arXiv:1612.00796 (2016)"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Knox, W.B., Stone, P.: Reinforcement learning from human reward: discounting in episodic tasks. In: Proceedings of the 21st IEEE International Symposium on Robot and Human Interactive Communication (RO-MAN), pp. 878\u2013885. IEEE Press (2012)","DOI":"10.1109\/ROMAN.2012.6343862"},{"key":"2_CR30","doi-asserted-by":"crossref","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518, 529\u2013533 (2015)","DOI":"10.1038\/nature14236"},{"issue":"5","key":"2_CR31","doi-asserted-by":"publisher","first-page":"1295","DOI":"10.1088\/0143-0807\/31\/5\/028","volume":"31","author":"J Morio","year":"2010","unstructured":"Morio, J., Pastel, R., Le Gland, F.: An overview of importance splitting for rare event simulation. Eur. J. Phys. 31(5), 1295 (2010)","journal-title":"Eur. J. Phys."},{"key":"2_CR32","unstructured":"Nazari, M., Oroojlooy, A., Snyder, L.V., Tak\u00e1c, M.: Reinforcement learning for solving the vehicle routing problem. In: Proceedings of the Annual Conference on Neural Information Processing Systems (NeurIPS), pp. 9861\u20139871 (2018)"},{"key":"2_CR33","unstructured":"Parker-Holder, J., et al.: Evolving curricula with regret-based environment design. In: Proceedings of the International Conference on Machine Learning (ICML), pp. 17473\u201317498. PMLR (2022)"},{"key":"2_CR34","unstructured":"Raileanu, R., Rockt\u00e4schel, T.: RIDE: rewarding impact-driven exploration for procedurally-generated environments. In: Proceedings of the 8th International Conference on Learning Representations (ICLR). OpenReview (2020)"},{"key":"2_CR35","unstructured":"Riedmiller, M.A., et al.: Learning by playing solving sparse reward tasks from scratch. In: Proceedings of the 35th International Conference on Machine Learning (ICML), pp. 4341\u20134350. PMLR (2018)"},{"issue":"19","key":"2_CR36","doi-asserted-by":"publisher","first-page":"70","DOI":"10.2352\/ISSN.2470-1173.2017.19.AVM-023","volume":"2017","author":"AE Sallab","year":"2017","unstructured":"Sallab, A.E., Abdou, M., Perot, E., Yogamani, S.: Deep reinforcement learning framework for autonomous driving. Electron. Imaging 2017(19), 70\u201376 (2017)","journal-title":"Electron. Imaging"},{"key":"2_CR37","unstructured":"Schaul, T., Quan, J., Antonoglou, I., Silver, D.: Prioritized experience replay. In: Proceedings of the 4th International Conference on Learning Representations (ICLR) (2016)"},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Schwartz, A.: A reinforcement learning method for maximizing undiscounted rewards. In: Proceedings of the 10th International Conference on Machine Learning (ICML), pp. 298\u2013305. Morgan Kaufmann (1993)","DOI":"10.1016\/B978-1-55860-307-3.50045-9"},{"key":"2_CR39","doi-asserted-by":"crossref","unstructured":"Silver, D., et al.: Mastering the game of go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2016)","DOI":"10.1038\/nature16961"},{"key":"2_CR40","doi-asserted-by":"crossref","unstructured":"Silver, D., et al.: A General reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419), 1140\u20131144 (2018)","DOI":"10.1126\/science.aar6404"},{"key":"2_CR41","doi-asserted-by":"crossref","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354\u2013359 (2017)","DOI":"10.1038\/nature24270"},{"key":"2_CR42","unstructured":"Stooke, A., Abbeel, P.: rlpyt: a research code base for deep reinforcement learning in PyTorch. arXiv preprint arXiv:1909.01500 (2019)"},{"key":"2_CR43","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning - An Introduction, Adaptive Computation and Machine Learning. MIT Press (1998)"}],"container-title":["Lecture Notes in Computer Science","Principles of Verification: Cycling the Probabilistic Landscape"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-75778-5_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,17]],"date-time":"2024-11-17T08:02:19Z","timestamp":1731830539000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-75778-5_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,18]]},"ISBN":["9783031757778","9783031757785"],"references-count":43,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-75778-5_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,18]]},"assertion":[{"value":"18 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}