{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T16:03:12Z","timestamp":1781280192658,"version":"3.54.1"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030634605","type":"print"},{"value":"9783030634612","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-63461-2_1","type":"book-chapter","created":{"date-parts":[[2020,11,13]],"date-time":"2020-11-13T16:03:28Z","timestamp":1605283408000},"page":"3-21","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Formal Policy Synthesis for Continuous-State Systems via Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Milad","family":"Kazemi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sadegh","family":"Soudjani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,11,13]]},"reference":[{"key":"1_CR1","volume-title":"Principles of Model Checking","author":"C Baier","year":"2008","unstructured":"Baier, C., Katoen, J.P.: Principles of Model Checking. MIT Press, Cambridge (2008)"},{"key":"1_CR2","volume-title":"Stochastic Optimal Control: The Discrete-Time Case","author":"D Bertsekas","year":"1996","unstructured":"Bertsekas, D., Shreve, S.: Stochastic Optimal Control: The Discrete-Time Case. Athena Scientific, Nashua (1996)"},{"key":"1_CR3","doi-asserted-by":"crossref","unstructured":"Bozkurt, A.K., Wang, Y., Zavlanos, M.M., Pajic, M.: Control synthesis from linear temporal logic specifications using model-free reinforcement learning. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 10349\u201310355. IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196796"},{"key":"1_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/978-3-319-11936-6_8","volume-title":"Automated Technology for Verification and Analysis","author":"T Br\u00e1zdil","year":"2014","unstructured":"Br\u00e1zdil, T., et al.: Verification of Markov decision processes using learning algorithms. In: Cassez, F., Raskin, J.-F. (eds.) ATVA 2014. LNCS, vol. 8837, pp. 98\u2013114. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-11936-6_8"},{"issue":"4","key":"1_CR5","doi-asserted-by":"publisher","first-page":"857","DOI":"10.1145\/210332.210339","volume":"42","author":"C Courcoubetis","year":"1995","unstructured":"Courcoubetis, C., Yannakakis, M.: The complexity of probabilistic verification. J. ACM (JACM) 42(4), 857\u2013907 (1995)","journal-title":"J. ACM (JACM)"},{"key":"1_CR6","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-6746-2","volume-title":"Controlled Markov Processes","author":"EB Dynkin","year":"1979","unstructured":"Dynkin, E.B., Yushkevich, A.A.: Controlled Markov Processes, vol. 235. Springer, New York (1979)"},{"key":"1_CR7","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.dam.2018.05.038","volume":"251","author":"J Flesch","year":"2018","unstructured":"Flesch, J., Predtetchinski, A., Sudderth, W.: Simplifying optimal strategies in limsup and liminf stochastic games. Discret. Appl. Math. 251, 40\u201356 (2018)","journal-title":"Discret. Appl. Math."},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Fu, J., Topcu, U.: Probably approximately correct MDP learning and control with temporal logic constraints. In: Proceedings of Robotics: Science and Systems (2014)","DOI":"10.15607\/RSS.2014.X.039"},{"issue":"1","key":"1_CR9","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a, J., Fern\u00e1ndez, F.: A comprehensive survey on safe reinforcement learning. J. Mach. Learn. Res. 16(1), 1437\u20131480 (2015)","journal-title":"J. Mach. Learn. Res."},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"Haesaert, S., Soudjani, S.: Robust dynamic programming for temporal logic control of stochastic systems. IEEE Trans. Autom. Control (2020)","DOI":"10.1109\/TAC.2020.3010490"},{"key":"1_CR11","unstructured":"Hahn, E.M., Li, G., Schewe, S., Turrini, A., Zhang, L.: Lazy probabilistic model checking without determinisation. In: International Conference on Concurrency Theory (CONCUR), pp. 354\u2013367 (2015)"},{"key":"1_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"395","DOI":"10.1007\/978-3-030-17462-0_27","volume-title":"Tools and Algorithms for the Construction and Analysis of Systems","author":"EM Hahn","year":"2019","unstructured":"Hahn, E.M., Perez, M., Schewe, S., Somenzi, F., Trivedi, A., Wojtczak, D.: Omega-regular objectives in model-free reinforcement learning. In: Vojnar, T., Zhang, L. (eds.) TACAS 2019. LNCS, vol. 11427, pp. 395\u2013412. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-17462-0_27"},{"key":"1_CR13","unstructured":"Hasanbeig, M., Abate, A., Kr\u00f6ning, D.: Logically-constrained reinforcement learning. arXiv preprint arXiv:1801.08099 (2018)"},{"key":"1_CR14","unstructured":"Hasanbeig, M., Abate, A., Kr\u00f6ning, D.: Logically-constrained neural fitted Q-iteration. In: Proceedings of the 18th International Conference on Autonomous Agents and MultiAgent Systems (AAMS), pp. 2012\u20132014 (2019)"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Hasanbeig, M., Kantaros, Y., Abate, A., Kroening, D., Pappas, G.J., Lee, I.: Reinforcement learning for temporal logic control synthesis with probabilistic satisfaction guarantees. In: IEEE Conference on Decision and Control (CDC), pp. 5338\u20135343. IEEE (2019)","DOI":"10.1109\/CDC40024.2019.9028919"},{"key":"1_CR16","doi-asserted-by":"crossref","unstructured":"Hasanbeig, M., Kroening, D., Abate, A.: Deep reinforcement learning with temporal logics. In: Formal Modeling and Analysis of Timed Systems, pp. 1\u201322 (2020)","DOI":"10.1007\/978-3-030-57628-8_1"},{"key":"1_CR17","series-title":"Stochastic Modelling and Applied Probability","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-0729-0","volume-title":"Discrete-Time Markov Control Processes: Basic Optimality Criteria","author":"O Hern\u00e1ndez-Lerma","year":"1996","unstructured":"Hern\u00e1ndez-Lerma, O., Lasserre, J.B.: Discrete-Time Markov Control Processes: Basic Optimality Criteria. Stochastic Modelling and Applied Probability, vol. 30. Springer, New York (1996)"},{"issue":"3","key":"1_CR18","doi-asserted-by":"publisher","first-page":"338","DOI":"10.1109\/5326.704563","volume":"28","author":"L Jouffe","year":"1998","unstructured":"Jouffe, L.: Fuzzy inference system learning by reinforcement methods. IEEE Trans. Syst. Man Cybern. 28(3), 338\u2013355 (1998)","journal-title":"IEEE Trans. Syst. Man Cybern."},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Kazemi, M., Soudjani, S.: Formal policy synthesis for continuous-space systems via reinforcement learning. arXiv:2005.01319 (2020)","DOI":"10.1007\/978-3-030-63461-2_1"},{"key":"1_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1007\/978-3-030-01090-4_34","volume-title":"Automated Technology for Verification and Analysis","author":"J K\u0159et\u00ednsk\u00fd","year":"2018","unstructured":"K\u0159et\u00ednsk\u00fd, J., Meggendorfer, T., Sickert, S.: Owl: a library for $$\\omega $$-words, automata, and LTL. In: Lahiri, S.K., Wang, C. (eds.) ATVA 2018. LNCS, vol. 11138, pp. 543\u2013550. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01090-4_34"},{"key":"1_CR21","doi-asserted-by":"crossref","unstructured":"Lavaei, A., Somenzi, F., Soudjani, S., Trivedi, A., Zamani, M.: Formal controller synthesis for continuous-space MDPs via model-free reinforcement learning. In: International Conference on Cyber-Physical Systems (ICCPS), pp. 98\u2013107 (2020)","DOI":"10.1109\/ICCPS48487.2020.00017"},{"key":"1_CR22","unstructured":"Lazaric, A., Restelli, M., Bonarini, A.: Reinforcement learning in continuous action spaces through sequential Monte Carlo methods. In: Advances in Neural Information Processing Systems, pp. 833\u2013840 (2008)"},{"issue":"2","key":"1_CR23","doi-asserted-by":"publisher","first-page":"861","DOI":"10.1214\/aop\/1176989271","volume":"21","author":"A Maitra","year":"1993","unstructured":"Maitra, A., Sudderth, W.: Borel stochastic games with lim sup payoff. Ann. Probab. 21(2), 861\u2013885 (1993)","journal-title":"Ann. Probab."},{"key":"1_CR24","doi-asserted-by":"crossref","unstructured":"Majumdar, R., Mallik, K., Soudjani, S.: Symbolic controller synthesis for B\u00fcchi specifications on stochastic systems. In: International Conference on Hybrid Systems: Computation and Control (HSCC). ACM, New York (2020)","DOI":"10.1145\/3365365.3382214"},{"key":"1_CR25","doi-asserted-by":"crossref","unstructured":"Mallik, K., Soudjani, S., Schmuck, A.K., Majumdar, R.: Compositional construction of finite state abstractions for stochastic control systems. In: Conference on Decision and Control (CDC), pp. 550\u2013557. IEEE (2017)","DOI":"10.1109\/CDC.2017.8263720"},{"key":"1_CR26","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning, vol. 48, pp. 1928\u20131937 (2016)"},{"issue":"2","key":"1_CR27","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1109\/72.279185","volume":"5","author":"SW Piche","year":"1994","unstructured":"Piche, S.W.: Steepest descent algorithms for neural network controllers and filters. IEEE Trans. Neural Netw. 5(2), 198\u2013212 (1994)","journal-title":"IEEE Trans. Neural Netw."},{"key":"1_CR28","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1146\/annurev-control-053018-023825","volume":"2","author":"B Recht","year":"2018","unstructured":"Recht, B.: A tour of reinforcement learning: the view from continuous control. Ann. Rev. Control Robot. Auton. Syst. 2, 253\u2013279 (2018)","journal-title":"Ann. Rev. Control Robot. Auton. Syst."},{"key":"1_CR29","doi-asserted-by":"crossref","unstructured":"Sadigh, D., Kim, E.S., Coogan, S., Sastry, S.S., Seshia, S.A.: A learning based approach to control synthesis of Markov decision processes for linear temporal logic specifications. In: Conference on Decision and Control, pp. 1091\u20131096 (2014)","DOI":"10.21236\/ADA623517"},{"key":"1_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"312","DOI":"10.1007\/978-3-319-41540-6_17","volume-title":"Computer Aided Verification","author":"S Sickert","year":"2016","unstructured":"Sickert, S., Esparza, J., Jaax, S., K\u0159et\u00ednsk\u00fd, J.: Limit-deterministic B\u00fcchi automata for linear temporal logic. In: Chaudhuri, S., Farzan, A. (eds.) CAV 2016. LNCS, vol. 9780, pp. 312\u2013332. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-41540-6_17"},{"key":"1_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.ic.2016.11.006","volume":"253","author":"I Tkachev","year":"2017","unstructured":"Tkachev, I., Mereacre, A., Katoen, J.P., Abate, A.: Quantitative model-checking of controlled discrete-time Markov processes. Inf. Comput. 253, 1\u201335 (2017)","journal-title":"Inf. Comput."},{"issue":"10","key":"1_CR32","doi-asserted-by":"publisher","first-page":"1329","DOI":"10.1177\/0278364915581505","volume":"34","author":"J Wang","year":"2015","unstructured":"Wang, J., Ding, X., Lahijanian, M., Paschalidis, I.C., Belta, C.A.: Temporal logic motion control using actor-critic methods. Int. J. Robot. Res. 34(10), 1329\u20131344 (2015)","journal-title":"Int. J. Robot. Res."}],"container-title":["Lecture Notes in Computer Science","Integrated Formal Methods"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-63461-2_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,14]],"date-time":"2021-04-14T05:44:10Z","timestamp":1618379050000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-63461-2_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030634605","9783030634612"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-63461-2_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"13 November 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"IFM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Integrated Formal Methods","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lugano","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Switzerland","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 November 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ifm2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ifm20.si.usi.ch\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"64","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"24","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"38% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"6,5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Due to the Corona pandemic this event was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}