{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:14:16Z","timestamp":1780766056515,"version":"3.54.1"},"reference-count":108,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3306070","type":"journal-article","created":{"date-parts":[[2023,8,17]],"date-time":"2023-08-17T17:32:21Z","timestamp":1692293541000},"page":"89188-89204","source":"Crossref","is-referenced-by-count":2,"title":["An Actor-Critic Framework for Online Control With Environment Stability Guarantee"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6184-3293","authenticated-orcid":false,"given":"Pavel","family":"Osinenko","sequence":"first","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8869-6422","authenticated-orcid":false,"given":"Grigory","family":"Yaremenko","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2936-6271","authenticated-orcid":false,"given":"Georgiy","family":"Malaniya","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anton","family":"Bolychev","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref57","article-title":"Plan online, learn offline: Efficient learning and exploration via model-based control","author":"lowrey","year":"2018","journal-title":"arXiv 1811 01848"},{"key":"ref56","first-page":"211","article-title":"Practical reinforcement learning for MPC: Learning from sparse objectives in under an hour on a real robot","volume":"120","author":"karnchanachari","year":"2020","journal-title":"Proc 2nd Conf Learn Dyn Control"},{"key":"ref59","first-page":"990","article-title":"Deep value model predictive control","volume":"100","author":"hoeller","year":"2020","journal-title":"Proc Conf Robot Learn"},{"key":"ref58","first-page":"8289","article-title":"Differentiable MPC for end-to-end planning and control","volume":"31","author":"amos","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619572"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.23919\/ECC.2019.8795816"},{"key":"ref55","article-title":"Safe exploration in reinforcement learning: Theory and applications in robotics","author":"berkenkamp","year":"2019"},{"key":"ref54","article-title":"Safe model-based reinforcement learning with stability guarantees","volume":"30","author":"berkenkamp","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2020.3024161"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3070252"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12107"},{"key":"ref45","first-page":"246","author":"platzer","year":"2009","journal-title":"Formal Methods and Software Engineering 11th International Conference on Formal Engineering Methods ICFEM 2009 Rio de Janeiro Brazil December 9&#x2013;12 2009 Proceedings"},{"key":"ref48","article-title":"Safe reinforcement learning using probabilistic shields","author":"k\u00f6nighofer","year":"2020","journal-title":"Proc 31st CONCUR Int Conf Concurrency Theory"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-61362-4_16"},{"key":"ref42","article-title":"Trial without error: Towards safe reinforcement learning via human intervention","author":"saunders","year":"2017","journal-title":"arXiv 1707 05173"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2023.3294187"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-71070-7_15"},{"key":"ref43","article-title":"Deductive stability proofs for ordinary differential equations","author":"tan","year":"2020","journal-title":"arXiv 2010 13096"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8593420"},{"key":"ref8","article-title":"Deep reinforcement learning for real autonomous mobile robot navigation in indoor environments","author":"surmann","year":"2020","journal-title":"arXiv 2005 13857"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386025"},{"key":"ref9","article-title":"Solving Rubik&#x2019;s cube with a robot hand","author":"akkaya","year":"2019","journal-title":"arXiv 1910 07113"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1049\/PBCE081E"},{"key":"ref3","volume":"17","author":"lewis","year":"2013","journal-title":"Reinforcement Learning and Approximate Dynamic Programming for Feedback Control"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2012.325"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487156"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2012.09.019"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysconle.2016.12.003"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2015.2414811"},{"key":"ref35","article-title":"Continuous-time model-based reinforcement learning","author":"y?ld?z","year":"2021","journal-title":"arXiv 2102 04764"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2014.10.128"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2008.926614"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1162\/089976600300015961"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1016\/0005-1098(89)90002-2"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/MCAS.2009.933854"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.conengprac.2011.12.004"},{"key":"ref32","volume":"20","author":"borrelli","year":"2011","journal-title":"Predictive Control for Linear and Hybrid Systems"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2281663"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2014.2314612"},{"key":"ref24","first-page":"5057","article-title":"True online temporal-difference learning","volume":"17","author":"van seijen","year":"2016","journal-title":"J Mach Learn Res"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.1997.650692"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-012-5280-0"},{"key":"ref25","first-page":"1377","article-title":"Temporal-difference networks","author":"sutton","year":"2005","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2018.8490427"},{"key":"ref22","first-page":"3948","article-title":"Asymptotically efficient off-policy evaluation for tabular reinforcement learning","author":"yin","year":"2020","journal-title":"Proc Int Conf Artif Intell Statist"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-018-1241-z"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2012.02.013"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2010.01.001"},{"key":"ref29","article-title":"Reinforcement learning for board games: The temporal difference algorithm","author":"konen","year":"2015"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1613\/jair.806"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","article-title":"Grandmaster level in StarCraft II using multi-agent reinforcement learning","volume":"575","author":"vinyals","year":"2019","journal-title":"Nature"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2006.282564"},{"key":"ref14","article-title":"A natural policy gradient","volume":"14","author":"kakade","year":"2001","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2020.XVI.088"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3045114"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref99","author":"khasminskii","year":"2011","journal-title":"Stochastic Stability of Differential Equations"},{"key":"ref10","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"ref17","first-page":"833","article-title":"Reinforcement learning in continuous action spaces through sequential Monte Carlo methods","volume":"20","author":"lazaric","year":"2007","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2006.311699"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.2202\/1553-779X.1066"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1613\/jair.5507"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2015.10.039"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2003.1239073"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1002\/rnc.670"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-78384-0"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/NICS.2018.8606873"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/TCSII.2018.2799625"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2899311"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2953613"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2019.2906694"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2019.2912828"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2018.2889377"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2011.2168538"},{"key":"ref81","article-title":"A Lyapunov-based approach to safe reinforcement learning","volume":"31","author":"chow","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref84","first-page":"8353","article-title":"Provably global convergence of actor-critic: A case for linear quadratic regulator with ergodic cost","author":"yang","year":"2019","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2015.2487972"},{"key":"ref80","first-page":"409","article-title":"Lyapunov-constrained action sets for reinforcement learning","volume":"1","author":"perkins","year":"2001","journal-title":"Proc ICML"},{"key":"ref79","first-page":"23","article-title":"Lyapunov design for safe reinforcement learning control","author":"perkins","year":"2002","journal-title":"Proc Safe Learn Agents Papers From AAAI Symp"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1080\/07362994.2019.1694416"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2020.109095"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6911(01)00164-5"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6911(98)00003-6"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.3182\/20120823-5-NL-3013.00037"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2022.3211986"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysconle.2014.08.002"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3154028"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1002\/rnc.3912"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2022.07.619"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2012.05.067"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/CDC45484.2021.9682838"},{"key":"ref2","author":"bertsekas","year":"2019","journal-title":"REINFORCEMENT LEARNING AND OPTIMAL CONTROL"},{"key":"ref1","author":"sutton","year":"2018","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.23919\/ECC51009.2020.9143704"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.23919\/ECC.2018.8550545"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1137\/070707853"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2008.927799"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9030185"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2023.3250032"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2018.10.175"},{"key":"ref64","article-title":"Dream to control: Learning behaviors by latent imagination","author":"hafner","year":"2020","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref63","first-page":"3741","article-title":"Fixed-horizon temporal difference methods for stable reinforcement learning","volume":"34","author":"asis","year":"2020","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2975727"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2020.2982585"},{"key":"ref60","article-title":"Infinite-horizon differentiable model predictive control","author":"east","year":"2020","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989324"},{"key":"ref61","article-title":"Learning human objectives by evaluating hypothetical behavior","author":"reddy","year":"2019","journal-title":"Proc Int Conf Mach Learn"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10223230.pdf?arnumber=10223230","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,18]],"date-time":"2023-09-18T18:11:11Z","timestamp":1695060671000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10223230\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":108,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3306070","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}