{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,18]],"date-time":"2026-04-18T09:39:32Z","timestamp":1776505172651,"version":"3.51.2"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2017,8,5]],"date-time":"2017-08-05T00:00:00Z","timestamp":1501891200000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2018,4]]},"DOI":"10.1007\/s10489-017-0999-8","type":"journal-article","created":{"date-parts":[[2017,8,5]],"date-time":"2017-08-05T01:28:35Z","timestamp":1501896515000},"page":"886-908","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["Verification and repair of control policies for safe reinforcement learning"],"prefix":"10.1007","volume":"48","author":[{"given":"Shashank","family":"Pathak","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luca","family":"Pulina","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9487-331X","authenticated-orcid":false,"given":"Armando","family":"Tacchella","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,8,5]]},"reference":[{"key":"999_CR1","doi-asserted-by":"crossref","unstructured":"Abrah\u00e1m E, Jansen N, Wimmer R, Katoen J, Becker B (2010) Dtmc model checking by scc reduction. In: 2010 7th international conference on the quantitative evaluation of systems (QEST). IEEE, pp 37\u201346","DOI":"10.1109\/QEST.2010.13"},{"key":"999_CR2","doi-asserted-by":"crossref","unstructured":"Aziz A, Singhal V, Balarin F, Brayton RK, Sangiovanni-Vincentell AL (1995) It usually works: the temporal logic of stochastic systems. In: Computer aided verification. Springer, pp 155\u2013165","DOI":"10.1007\/3-540-60045-0_48"},{"key":"999_CR3","unstructured":"Avriel M (2003) Nonlinear programming: analysis and methods. Courier Corporation"},{"key":"999_CR4","doi-asserted-by":"crossref","unstructured":"Bentivegna DC, Atkeson CG, Ude A, Cheng G (2004) Learning to act from observation and practice. Int J Human Robot 1(4)","DOI":"10.1142\/S0219843604000307"},{"key":"999_CR5","first-page":"1017","volume":"8","author":"A Barto","year":"1996","unstructured":"Barto A, Crites RH (1996) Improving elevator performance using reinforcement learning. Adv Neural Inf Process Syst 8:1017\u20131023","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1","key":"999_CR6","first-page":"94","volume":"11","author":"C Boutilier","year":"1999","unstructured":"Boutilier C, Dean T, Hanks S (1999) Decision-theoretic planning: structural assumptions and computational leverage. J Artif Intell Res 11(1):94","journal-title":"J Artif Intell Res"},{"issue":"1","key":"999_CR7","doi-asserted-by":"crossref","first-page":"57","DOI":"10.1016\/S0004-3702(99)00039-9","volume":"112","author":"F Buccafurri","year":"1999","unstructured":"Buccafurri F, Eiter T, Gottlob G, Leone N et al (1999) Enhancing model checking in verification by ai techniques. Artif Intell 112(1):57\u2013104","journal-title":"Artif Intell"},{"key":"999_CR8","doi-asserted-by":"crossref","unstructured":"Bartocci E, Grosu R, Katsaros P, Ramakrishnan C, Smolka S (2011) Model repair for probabilistic systems. Tools Algor Construct Anal Syst 326\u2013340","DOI":"10.1007\/978-3-642-19835-9_30"},{"key":"999_CR9","unstructured":"Ben-Israel A, Greville TNE (2003) Generalized inverses: theory and applications, vol 15. Springer Science & Business Media"},{"key":"999_CR10","doi-asserted-by":"crossref","unstructured":"Barrett L, Narayanan S (2008) Learning all optimal policies with multiple criteria. In: Proceedings of the 25th international conference on machine learning. ACM, pp 41\u201347","DOI":"10.1145\/1390156.1390162"},{"issue":"3","key":"999_CR11","doi-asserted-by":"crossref","first-page":"575","DOI":"10.1016\/j.compchemeng.2008.08.006","volume":"33","author":"LT Biegler","year":"2009","unstructured":"Biegler LT, Zavala VM (2009) Large-scale nonlinear programming using ipopt: an integrating framework for enterprise-wide dynamic optimization. Comput Chem Eng 33(3):575\u2013582","journal-title":"Comput Chem Eng"},{"key":"999_CR12","doi-asserted-by":"crossref","unstructured":"Cicala G, Khalili A, Metta G, Natale L, Pathak S, Pulina L, Tacchella A (2014) Engineering approaches and methods to verify software in autonomous systems. In: 13th international conference on intelligent autonomous systems (IAS-13)","DOI":"10.1007\/978-3-319-08338-4_121"},{"issue":"4","key":"999_CR13","doi-asserted-by":"crossref","first-page":"857","DOI":"10.1145\/210332.210339","volume":"42","author":"C Courcoubetis","year":"1995","unstructured":"Courcoubetis C, Yannakakis M (1995) The complexity of probabilistic verification. J ACM (JACM) 42(4):857\u2013907","journal-title":"J ACM (JACM)"},{"key":"999_CR14","doi-asserted-by":"crossref","unstructured":"Daws C (2005) Symbolic and parametric model checking of discrete-time Markov chains. In: Theoretical aspects of computing-ICTAC 2004. Springer, pp 280\u2013294","DOI":"10.1007\/978-3-540-31862-0_21"},{"key":"999_CR15","doi-asserted-by":"crossref","unstructured":"Filieri A, Ghezzi C, Tamburrelli G (2011) Run-time efficient probabilistic model checking. In: Proceedings of the 33rd international conference on software engineering. ACM, pp 341\u2013350","DOI":"10.1145\/1985793.1985840"},{"issue":"1","key":"999_CR16","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a J, Fern\u00e1ndez F (2015) A comprehensive survey on safe reinforcement learning. J Mach Learn Res 16(1):1437\u20131480","journal-title":"J Mach Learn Res"},{"key":"999_CR17","doi-asserted-by":"crossref","unstructured":"Ghallab M, Nau D, Traverso P (2004) Automated planning: theory & practice. Elsevier","DOI":"10.1016\/B978-155860856-6\/50021-1"},{"issue":"1","key":"999_CR18","doi-asserted-by":"crossref","first-page":"95","DOI":"10.1613\/jair.720","volume":"13","author":"DF Gordon","year":"2000","unstructured":"Gordon DF (2000) Asimovian adaptive agents. J Artif Intell Res 13(1):95\u2013153","journal-title":"J Artif Intell Res"},{"key":"999_CR19","unstructured":"Grinstead CM, Snell JL (1988) Introduction to probability. American Mathematical Soc. Chapter 11"},{"key":"999_CR20","unstructured":"Gillula JH, Tomlin CJ (2012) Guaranteed safe online learning via reachability: tracking a ground target using a quadrotor. In: ICRA, pp 2723\u20132730"},{"key":"999_CR21","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1613\/jair.1666","volume":"24","author":"P Geibel","year":"2005","unstructured":"Geibel P, Wysotzki F (2005) Risk-sensitive reinforcement learning applied to control under constraints. J Artif Intell Res 24:81\u2013108","journal-title":"J Artif Intell Res"},{"key":"999_CR22","doi-asserted-by":"crossref","unstructured":"Hahn EM, Hermanns H, Wachter B, Lijun Z (2010) PARAM: a model checker for parametric Markov models. In: Computer aided verification. Springer, pp 660\u2013664","DOI":"10.1007\/978-3-642-14295-6_56"},{"key":"999_CR23","doi-asserted-by":"crossref","unstructured":"Jansen N, \u00c1brah\u00e1m E, Volk M, Wimmer R, Katoen J-P, Becker B (2012) The comics tool\u2013computing minimal counterexamples for dtmcs. In: Automated technology for verification and analysis. Springer, pp 349\u2013353","DOI":"10.1007\/978-3-642-33386-6_27"},{"key":"999_CR24","doi-asserted-by":"crossref","unstructured":"Kwiatkowska M, Norman G, Parker D (2002) Prism: probabilistic symbolic model checker. In: Computer performance evaluation: modelling techniques and tools, pp 113\u2013140","DOI":"10.1007\/3-540-46029-2_13"},{"key":"999_CR25","doi-asserted-by":"crossref","unstructured":"Kwiatkowska M, Norman G, Parker D (2007) Stochastic model checking. Formal Methods Perform Eval 220\u2013270","DOI":"10.1007\/978-3-540-72522-0_6"},{"issue":"2","key":"999_CR26","doi-asserted-by":"crossref","first-page":"90","DOI":"10.1016\/j.peva.2010.04.001","volume":"68","author":"JP Katoen","year":"2011","unstructured":"Katoen JP, Zapreev IS, Hahn EM, Hermanns H, Jansen DN (2011) The ins and outs of the probabilistic model checker mrmc. Perform Eval 68(2):90\u2013104","journal-title":"Perform Eval"},{"key":"999_CR27","doi-asserted-by":"crossref","unstructured":"Leofante F, Vuotto S, A\u0307braha\u0307m E, Tacchella A, Jansen N (2016) Combining static and runtime methods to achieve safe standing-up for humanoid robots. In: Leveraging applications of formal methods, verification and validation: foundational techniques - 7th international symposium, ISoLA 2016, Imperial, Corfu, Greece, October 10-14, 2016, Proceedings, Part I, pp 496\u2013514","DOI":"10.1007\/978-3-319-47166-2_34"},{"key":"999_CR28","doi-asserted-by":"crossref","unstructured":"Morimoto J, Doya K (1998) Reinforcement learning of dynamic motor sequence Learning to stand up. In: Proceedings of the 1998 IEEE\/RSJ international conference on intelligent robots and systems, vol 3, pp 1721\u20131726","DOI":"10.1109\/IROS.1998.724846"},{"issue":"1","key":"999_CR29","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1016\/S0921-8890(01)00113-0","volume":"36","author":"J Morimoto","year":"2001","unstructured":"Morimoto J, Doya K (2001) Acquisition of stand-up behavior by a real robot using hierarchical reinforcement learning. Robot Auton Syst 36(1):37\u201351","journal-title":"Robot Auton Syst"},{"key":"999_CR30","doi-asserted-by":"crossref","unstructured":"Metta G, Natale L, Nori F, Sandini G, Vernon D, Fadiga L, von Hofsten C, Rosander K, Lopes M, Santos-Victor J et al (2010) The iCub humanoid robot: an open-systems platform for research in cognitive development. Neural networks: the official journal of the international neural network society","DOI":"10.1016\/j.neunet.2010.08.010"},{"key":"999_CR31","doi-asserted-by":"crossref","unstructured":"Metta G, Natale L, Pathak S, Pulina L, Tacchella A (2010) Safe and effective learning: a case study. In: 2010 IEEE international conference on robotics and automation, pp 4809\u20134814","DOI":"10.1109\/ROBOT.2010.5509892"},{"key":"999_CR32","unstructured":"Metta G, Pathak S, Pulina L, Tacchella A (2013) Ensuring safety of policies learned by reinforcement: reaching objects in the presence of obstacles with the iCub. In: IEEE\/RSJ international conference on intelligent robots and systems, pp 170\u2013175"},{"key":"999_CR33","doi-asserted-by":"crossref","unstructured":"Ng A, Coates A, Diel M, Ganapathi V, Schulte J, Tse B, Berger E, Liang E (2006) Autonomous inverted helicopter flight via reinforcement learning. Exper Robot IX 363\u2013372","DOI":"10.1007\/11552246_35"},{"key":"999_CR34","doi-asserted-by":"crossref","unstructured":"Natarajan S, Tadepalli P (2005) Dynamic preferences in multi-criteria reinforcement learning. In: Proceedings of the 22nd international conference on machine learning. ACM, pp 601\u2013608","DOI":"10.1145\/1102351.1102427"},{"key":"999_CR35","doi-asserted-by":"crossref","unstructured":"Pathak S, Abraham E, Jansen N, Tacchella A, Katoen JP (2015) A greedy approach for the efficient repair of stochastic models. In: Proc. NFM\u201915, volume 9058 of LNCS, pp 295\u2013309","DOI":"10.1007\/978-3-319-17524-9_21"},{"key":"999_CR36","first-page":"803","volume":"3","author":"TJ Perkins","year":"2003","unstructured":"Perkins TJ, Barto AG (2003) Lyapunov design for safe reinforcement learning. J Mach Learn Res 3:803\u2013832","journal-title":"J Mach Learn Res"},{"key":"999_CR37","doi-asserted-by":"crossref","unstructured":"Pathak S, Metta G, Tacchella A (2014) Is verification a requisite for safe adaptive robots? In: 2014 IEEE international conference on systems, man and cybernetics","DOI":"10.1109\/SMC.2014.6974453"},{"key":"999_CR38","doi-asserted-by":"crossref","unstructured":"Pathak S, Pulina L, Tacchella A (2015) Probabilistic model checking tools for verification of robot control policies. AI Commun. To appear","DOI":"10.3233\/AIC-150689"},{"key":"999_CR39","unstructured":"Puterman ML (2009) Markov decision processes: discrete stochastic dynamic programming, vol 414. Wiley"},{"key":"999_CR40","unstructured":"Rummery GA, Niranjan M (1994) On-line Q-learning using connectionist. University of Cambridge Department of Engineering"},{"key":"999_CR41","unstructured":"Russell S, Norvig P (2003) Artificial intelligence: a modern approach, 2nd edn. Prentice Hall"},{"key":"999_CR42","doi-asserted-by":"crossref","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning \u2013 an introduction. MIT Press","DOI":"10.1016\/S1474-6670(17)38315-5"},{"issue":"3","key":"999_CR43","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1023\/A:1007678930559","volume":"38","author":"S Singh","year":"2000","unstructured":"Singh S, Jaakkola T, Littman ML, Szepesv\u00e1ri C (2000) Convergence results for single-step on-policy reinforcement-learning algorithms. Mach Learn 38(3):287\u2013308","journal-title":"Mach Learn"},{"key":"999_CR44","unstructured":"Smith DJ, Simpson KGL (2004) Functional safety \u2013 a straightforward guide to applying IEC 61505 and related standards, 2nd edn. Elsevier"},{"issue":"3","key":"999_CR45","doi-asserted-by":"crossref","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro G (1995) Temporal difference learning and td-gammon. Commun ACM 38(3):58\u201368","journal-title":"Commun ACM"},{"issue":"1","key":"999_CR46","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1007\/s10107-004-0559-y","volume":"106","author":"A W\u00e4chter","year":"2006","unstructured":"W\u00e4chter A, Biegler LT (2006) On the implementation of an interior-point filter line-search algorithm for large-scale nonlinear programming. Math Program 106(1):25\u201357","journal-title":"Math Program"},{"issue":"3","key":"999_CR47","first-page":"279","volume":"8","author":"CJCH Watkins","year":"1992","unstructured":"Watkins CJCH, Dayan P (1992) Q-learning. Mach Learn 8(3):279\u2013292","journal-title":"Mach Learn"},{"key":"999_CR48","unstructured":"Weld D, Etzioni O (1994) The first law of robotics (a call to arms). In: Proceedings of the 12th national conference on artificial intelligence (AAAI-94), pp 1042\u20131047"},{"key":"999_CR49","unstructured":"Zhang W, Dietterich TG (1995) A reinforcement learning approach to job-shop scheduling. In: IJCAI, vol 95. Citeseer, pp 1114\u20131120"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10489-017-0999-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-017-0999-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-017-0999-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,10,1]],"date-time":"2019-10-01T22:09:08Z","timestamp":1569967748000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10489-017-0999-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,8,5]]},"references-count":49,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2018,4]]}},"alternative-id":["999"],"URL":"https:\/\/doi.org\/10.1007\/s10489-017-0999-8","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,8,5]]}}}