{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T02:14:30Z","timestamp":1771467270591,"version":"3.50.1"},"reference-count":52,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,8,28]],"date-time":"2024-08-28T00:00:00Z","timestamp":1724803200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,8,28]],"date-time":"2024-08-28T00:00:00Z","timestamp":1724803200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,8,28]]},"DOI":"10.1109\/case59546.2024.10711661","type":"proceedings-article","created":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T17:40:16Z","timestamp":1729705216000},"page":"2441-2448","source":"Crossref","is-referenced-by-count":1,"title":["Safe Value Functions: Learned Critics as Hard Safety Constraints"],"prefix":"10.1109","author":[{"given":"Daniel C.H.","family":"Tan","sequence":"first","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Robert","family":"McCarthy","sequence":"additional","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fernando","family":"Acero","sequence":"additional","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andromachi Maria","family":"Delfaki","sequence":"additional","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhibin","family":"Li","sequence":"additional","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dimitrios","family":"Kanoulas","sequence":"additional","affiliation":[{"name":"University College London,Department of Computer Science,London,UK,WC1E 6BT"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Playing atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref2","article-title":"Continuous control with deep reinforcement learning","volume-title":"4th International Conference on Learning Representations, ICLR 2016 - Conference Track Proceedings","author":"Lillicrap"},{"key":"ref3","first-page":"583","article-title":"Highly accurate protein structure prediction with alphafold","volume-title":"Nature 2021 596:7873","volume":"596","author":"Jumper","year":"2021"},{"key":"ref4","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2014.7040372"},{"issue":"12","key":"ref6","first-page":"462","article-title":"CONSTRUCTIVE SAFETY USING CONTROL BARRIER FUNCTIONS","volume-title":"IFAC Proceedings Volumes","volume":"40","author":"Wieland"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1115\/DSCC2014-6048"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/tro.2022.3232542"},{"key":"ref9","article-title":"Neural Lyapunov Control","author":"Chang","year":"2022"},{"key":"ref10","first-page":"112","article-title":"Learning Barrier Functions for Constrained Motion Planning with Dynamical Systems","volume-title":"2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","author":"Saveriano"},{"key":"ref11","doi-asserted-by":"crossref","DOI":"10.1109\/IROS45743.2020.9341190","article-title":"Synthesis of Control Barrier Functions Using a Supervised Machine Learning Approach","author":"Srinivasan","year":"2020"},{"key":"ref12","doi-asserted-by":"crossref","DOI":"10.1109\/IROS51168.2021.9636468","article-title":"Model-based Constrained Reinforcement Learning using Generalized Control Barrier Function","author":"Ma","year":"2021"},{"key":"ref13","article-title":"Learning Safe Multi-Agent Control with Decentralized Neural Barrier Certificates","author":"Qin","year":"2021"},{"key":"ref14","volume-title":"Constrained Markov Decision Processes","author":"Altman","year":"1999"},{"key":"ref15","article-title":"Constrained policy optimization","author":"Achiam","year":"2017"},{"key":"ref16","article-title":"Your Value Function is a Control Barrier Function (Outstanding Paper Award)","volume-title":"International Conference on Machine Learning, Workshop on Formal Verification and Machine Learning","author":"Tan"},{"key":"ref17","article-title":"Safe reinforcement learning by imagining the near future","author":"Thomas","year":"2022"},{"issue":"5","key":"ref18","first-page":"2743","article-title":"Safe value functions","volume-title":"IEEE Transactions on Automatic Control","volume":"68","author":"Massiani","year":"2023"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"ref20","first-page":"19","article-title":"Combining model-based design and model-free policy optimization to learn safe, stabilizing controllers","volume-title":"IFAC-PapersOnLine","volume":"54","author":"Westenbroek"},{"key":"ref21","article-title":"Lyapunov design for robust and efficient robotic reinforcement learning","author":"Westenbroek","year":"2022"},{"key":"ref22","article-title":"A barrier-lyapunov actor-critic reinforcement learning approach for safe and stable control","author":"Zhao","year":"2023"},{"key":"ref23","article-title":"Feasible policy iteration","author":"Yang","year":"2023"},{"key":"ref24","first-page":"1357","article-title":"Robot reinforcement learning on the constraint manifold","volume-title":"Proceedings of the 5th Conference on Robot Learning","volume":"164","author":"Liu"},{"key":"ref25","article-title":"Safe exploration in continuous action spaces","author":"Dalal","year":"2018"},{"key":"ref26","article-title":"Safe control under input limits with neural control barrier functions","author":"Liu","year":"2022"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-042920-020211"},{"key":"ref28","first-page":"773","article-title":"Formal synthesis of lyapunov neural networks","volume-title":"IEEE Control Systems Letters","volume":"5","author":"Abate","year":"2021"},{"key":"ref29","article-title":"The lyapunov neural network: Adaptive stability certification for safe learning of dynamical systems","author":"Richards","year":"2018"},{"key":"ref30","article-title":"Learning stable deep dynamics models","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Manek","year":"2020"},{"key":"ref31","first-page":"2091","article-title":"Lyapunov-net: A deep neural network architecture for lyapunov function approximation","volume-title":"Proceedings of the IEEE Conference on Decision and Control","volume":"2022-December","author":"Gaby"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6911(01)00110-4"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CDC42340.2020.9304201"},{"key":"ref34","first-page":"1928","article-title":"Sablas: Learning safe control for black-box dynamical systems","volume-title":"IEEE Robotics and Automation Letters","volume":"7","author":"Qin","year":"2022"},{"key":"ref35","article-title":"Reward constrained policy optimization","author":"Tessler","year":"2018"},{"key":"ref36","article-title":"Responsive safety in reinforcement learning by pid lagrangian methods","volume-title":"37th International Conference on Machine Learning","author":"Stooke"},{"key":"ref37","article-title":"A lyapunov-based approach to safe reinforcement learning","author":"Chow","year":"2018"},{"key":"ref38","article-title":"Openai gym","author":"Brockman","year":"2016"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref40","article-title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","author":"Fu","year":"2021"},{"key":"ref41","article-title":"Conservative Q-Learning for Offline Reinforcement Learning","author":"Kumar","year":"2020"},{"key":"ref42","article-title":"CORL: Research-oriented Deep Offline Reinforcement Learning Library","author":"Tarasov","year":"2023"},{"key":"ref43","article-title":"Proximal Policy Optimization Algorithms","author":"Schulman","year":"2017"},{"key":"ref44","article-title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","author":"Haarnoja","year":"2018"},{"key":"ref45","first-page":"1","article-title":"Cleanrl: High-quality single-file implementations of deep reinforcement learning algorithms","volume-title":"Journal of Machine Learning Research","volume":"23","author":"Huang","year":"2022"},{"key":"ref46","article-title":"Benchmarking Safe Exploration in Deep Reinforcement Learning","author":"Achiam","year":"2019"},{"key":"ref47","article-title":"Omnisafe: An infrastructure for accelerating safe reinforcement learning research","author":"Ji","year":"2023"},{"key":"ref48","article-title":"A Minimalist Approach to Offline Reinforcement Learning","author":"Fujimoto","year":"2021"},{"key":"ref49","article-title":"Almost Lyapunov Functions for Nonlinear Systems","author":"Liu","year":"2018"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9560886"},{"key":"ref51","first-page":"1904","article-title":"Learning safe, generalizable perception-based hybrid control with certificates","volume-title":"IEEE Robotics and Automation Letters","volume":"7","author":"Dawson","year":"2022"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.2174\/1573399812666160613113556"}],"event":{"name":"2024 IEEE 20th International Conference on Automation Science and Engineering (CASE)","location":"Bari, Italy","start":{"date-parts":[[2024,8,28]]},"end":{"date-parts":[[2024,9,1]]}},"container-title":["2024 IEEE 20th International Conference on Automation Science and Engineering (CASE)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10711304\/10711288\/10711661.pdf?arnumber=10711661","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T01:52:47Z","timestamp":1732672367000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10711661\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,28]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/case59546.2024.10711661","relation":{},"subject":[],"published":{"date-parts":[[2024,8,28]]}}}