{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T20:17:46Z","timestamp":1783455466513,"version":"3.55.0"},"reference-count":60,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tpami.2026.3674995","type":"journal-article","created":{"date-parts":[[2026,3,17]],"date-time":"2026-03-17T20:23:57Z","timestamp":1773779037000},"page":"9142-9155","source":"Crossref","is-referenced-by-count":0,"title":["Mirror Descent Safe Policy Optimization for Reinforcement Learning Agents"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6845-1711","authenticated-orcid":false,"given":"Renzhi","family":"Lu","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence and Automation, Key Laboratory of Image Processing and Intelligent Control, Engineering Research Center of Autonomous Intelligent Unmanned Systems, Chinese Ministry of Education, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ning","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and Automation, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingqing","family":"Xiong","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and Automation, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1607-3629","authenticated-orcid":false,"given":"Yifang","family":"Shi","sequence":"additional","affiliation":[{"name":"School of Automation, Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7153-9703","authenticated-orcid":false,"given":"Dongrui","family":"Wu","sequence":"additional","affiliation":[{"name":"Key Laboratory of the Ministry of Education for Image Processing and Intelligent Control, School of Artificial Intelligence and Automation, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4090-8497","authenticated-orcid":false,"given":"Tao","family":"Yang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Synthetical Automation for Process Industries, Northeastern University, Shenyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1100-0631","authenticated-orcid":false,"given":"Yaochu","family":"Jin","sequence":"additional","affiliation":[{"name":"School of Engineering, Westlake University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7137-4136","authenticated-orcid":false,"given":"Lihua","family":"Xie","sequence":"additional","affiliation":[{"name":"School of Electrical and Electronics Engineering, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-28487-2"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.34133\/research.1123"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3277206"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-025-66009-y"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3326851"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-024-00931-6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TFUZZ.2025.3599537"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2022.3183802"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-018-0009-9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2021.3104596"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-021-25874-z"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.jai.2023.11.001"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2024.3424529"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.34133\/research.0064"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.jai.2023.100018"},{"key":"ref16","article-title":"Playing atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref18","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"issue":"39","key":"ref20","first-page":"1","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989385"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3474289"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.34133\/research.0299"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.2352\/ISSN.2470-1173.2017.19.AVM-023"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/IVS.2018.8500556"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8461233"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.apenergy.2020.115473"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3457538"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"ref30","article-title":"Exploration-exploitation in constrained mdps","author":"Efroni","year":"2020"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2014.2309262"},{"key":"ref32","first-page":"9797","article-title":"Safe reinforcement learning in constrained markov decision processes","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wachi","year":"2020"},{"key":"ref33","first-page":"803","article-title":"Lyapunov design for safe reinforcement learning","volume":"3","author":"Perkins","year":"2002","journal-title":"J. Mach. Learn. Res."},{"key":"ref34","first-page":"908","article-title":"Safe model-based reinforcement learning with stability guarantees","volume-title":"Proc. 31st Int. Conf. Neural Inf. Process. Syst.","author":"Berkenkamp","year":"2017"},{"key":"ref35","article-title":"Lyapunov-based safe policy optimization for continuous control","author":"Chow","year":"2019"},{"key":"ref36","article-title":"Safe exploration for constrained reinforcement learning with provable guarantees","author":"Bura","year":"2021"},{"key":"ref37","first-page":"13859","article-title":"Safe reinforcement learning by imagining the near future","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Thomas","year":"2021"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i5.20478"},{"issue":"167","key":"ref39","first-page":"1","article-title":"Risk-constrained reinforcement learning with percentile risk criteria","volume":"18","author":"Chow","year":"2018","journal-title":"J. Mach. Learn. Res."},{"key":"ref40","article-title":"Reward constrained policy optimization","author":"Tessler","year":"2018"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2024.3514128"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3051456"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.3044196"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3265358"},{"key":"ref45","article-title":"Trust region policy optimization","author":"Schulman","year":"2015"},{"key":"ref46","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref47","first-page":"22","article-title":"Constrained policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Achiam","year":"2017"},{"key":"ref48","article-title":"Projection-based constrained policy optimization","author":"Yang","year":"2020"},{"key":"ref49","first-page":"15338","article-title":"First order constrained optimization in policy space","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Zhang","year":"2020"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0662"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/520"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2015.2388583"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2023.3244995"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/j.rico.2021.100048"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9992419"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2424"},{"key":"ref57","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0831"},{"issue":"285","key":"ref59","first-page":"1","article-title":"Omnisafe: An infrastructure for accelerating safe reinforcement learning research","volume":"25","author":"Ji","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref60","first-page":"267","article-title":"Approximately optimal approximate reinforcement learning","volume-title":"Proc. 19th Int. Conf. Mach. Learn.","author":"Kakade","year":"2002"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11595778\/11435918.pdf?arnumber=11435918","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T19:44:54Z","timestamp":1783453494000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11435918\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":60,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3674995","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}