{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T17:34:14Z","timestamp":1783445654745,"version":"3.54.6"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,12,5]],"date-time":"2023-12-05T00:00:00Z","timestamp":1701734400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,12,5]],"date-time":"2023-12-05T00:00:00Z","timestamp":1701734400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,12,5]]},"DOI":"10.1109\/icar58858.2023.10406465","type":"proceedings-article","created":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T13:27:00Z","timestamp":1706794020000},"page":"325-331","source":"Crossref","is-referenced-by-count":6,"title":["Bypassing the Simulation-to-Reality Gap: Online Reinforcement Learning Using a Supervisor"],"prefix":"10.1109","author":[{"given":"Benjamin David","family":"Evans","sequence":"first","affiliation":[{"name":"Stellenbosch University,Department of Electrical and Electronic Engineering,South Africa"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Johannes","family":"Betz","sequence":"additional","affiliation":[{"name":"Technical University of Munich,Professorship Autonomous Vehicle Systems,Munich,Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongrui","family":"Zheng","sequence":"additional","affiliation":[{"name":"University of Pennsylvania,Department of Electrical and Systems Engineering,Philadelphia,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Herman A.","family":"Engelbrecht","sequence":"additional","affiliation":[{"name":"Stellenbosch University,Department of Electrical and Electronic Engineering,South Africa"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rahul","family":"Mangharam","sequence":"additional","affiliation":[{"name":"University of Pennsylvania,Department of Electrical and Systems Engineering,Philadelphia,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hendrik W.","family":"Jordaan","sequence":"additional","affiliation":[{"name":"Stellenbosch University,Department of Electrical and Electronic Engineering,South Africa"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","author":"Shalev-Shwartz","year":"2016","journal-title":"Safe, multi-agent, reinforcement learning for autonomous driving"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"ref3","author":"Zhu","year":"2020","journal-title":"The ingredients of real-world robotic reinforcement learning"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-021-05961-4"},{"issue":"1","key":"ref5","first-page":"1437","article-title":"A comprehensive survey on safe reinforcement learning","volume":"16","author":"Garcia","year":"2015","journal-title":"Journal of Machine Learning Research"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2019.2928820"},{"key":"ref7","author":"Mirowski","year":"2016","journal-title":"Learning to navigate in complex environments"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-04357-7"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC.2019.8917387"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3064284"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460934"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2967299"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9811650"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3097345"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9562079"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3061627"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC45102.2020.9294262"},{"key":"ref18","author":"Cai","year":"2022","journal-title":"Overcoming exploration: Deep reinforcement learning in complex environments from temporal logic specifications"},{"issue":"1","key":"ref19","article-title":"Safe reinforcement learning via shielding","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"32","author":"Alshiekh"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8206234"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3152313"},{"key":"ref22","first-page":"708","article-title":"Learning for safety-critical control with control barrier functions","volume-title":"Learning for Dynamics and Control","author":"Taylor","year":"2020"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2018.8430770"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/icra.2019.8793742"},{"key":"ref26","first-page":"2067","article-title":"Trial without error: Towards safe reinforcement learning via human intervention","volume-title":"Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems","author":"Saunders","year":"2018"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CCNC49033.2022.9700730"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICAA52185.2022.00010"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/OJITS.2022.3181510"},{"key":"ref30","volume":"abs\/1705.01167","author":"Walsh","year":"2017","journal-title":"Cddt: Fast approximate 2d ray casting for accelerated localization"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/IVS.2017.7995802"},{"key":"ref32","volume-title":"Implementation of the pure pursuit path tracking algorithm","author":"Coulter","year":"1992"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1080\/00423114.2019.1631455"},{"key":"ref34","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"International conference on machine learning","author":"Fujimoto","year":"2018"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2017.2772903"},{"key":"ref36","article-title":"F1tenth: An open-source evaluation environment for continuous control and reinforcement learning","volume-title":"Proceedings of Machine Learning Research","volume":"123","author":"OKelly","year":"2020"}],"event":{"name":"2023 21st International Conference on Advanced Robotics (ICAR)","location":"Abu Dhabi, United Arab Emirates","start":{"date-parts":[[2023,12,5]]},"end":{"date-parts":[[2023,12,8]]}},"container-title":["2023 21st International Conference on Advanced Robotics (ICAR)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10406297\/10406219\/10406465.pdf?arnumber=10406465","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,2]],"date-time":"2024-02-02T13:26:45Z","timestamp":1706880405000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10406465\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,5]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/icar58858.2023.10406465","relation":{},"subject":[],"published":{"date-parts":[[2023,12,5]]}}}