{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T16:01:02Z","timestamp":1783440062729,"version":"3.54.6"},"reference-count":74,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T00:00:00Z","timestamp":1696118400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T00:00:00Z","timestamp":1696118400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T00:00:00Z","timestamp":1696118400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Australian Research Council Project","award":["FL-170100117"],"award-info":[{"award-number":["FL-170100117"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1109\/tpami.2023.3287908","type":"journal-article","created":{"date-parts":[[2023,6,20]],"date-time":"2023-06-20T17:27:38Z","timestamp":1687282058000},"page":"12236-12249","source":"Crossref","is-referenced-by-count":4,"title":["Prescribed Safety Performance Imitation Learning From a Single Expert Dataset"],"prefix":"10.1109","volume":"45","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8360-5149","authenticated-orcid":false,"given":"Zhihao","family":"Cheng","sequence":"first","affiliation":[{"name":"Sydney AI Centre and the School of Computer Science, in the Faculty of Engineering, The University of Sydney, Darlington, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5659-3464","authenticated-orcid":false,"given":"Li","family":"Shen","sequence":"additional","affiliation":[{"name":"JD Explore Academy, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9534-1999","authenticated-orcid":false,"given":"Miaoxi","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Computer Science, the National Engineering Research Center for Multimedia Software, the Institute of Artificial Intelligence, and the Hubei Key Laboratory of Multimedia and Network Communication Engineering, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0212-8329","authenticated-orcid":false,"given":"Jiaxian","family":"Guo","sequence":"additional","affiliation":[{"name":"Sydney AI Centre and the School of Computer Science, in the Faculty of Engineering, The University of Sydney, Darlington, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6745-286X","authenticated-orcid":false,"given":"Meng","family":"Fang","sequence":"additional","affiliation":[{"name":"University of Liverpool, Liverpool, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8128-2788","authenticated-orcid":false,"given":"Liu","family":"Liu","sequence":"additional","affiliation":[{"name":"Sydney AI Centre and the School of Computer Science, in the Faculty of Engineering, The University of Sydney, Darlington, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0059-8458","authenticated-orcid":false,"given":"Bo","family":"Du","sequence":"additional","affiliation":[{"name":"School of Computer Science, the National Engineering Research Center for Multimedia Software, the Institute of Artificial Intelligence, and the Hubei Key Laboratory of Multimedia and Network Communication Engineering, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7225-5449","authenticated-orcid":false,"given":"Dacheng","family":"Tao","sequence":"additional","affiliation":[{"name":"Sydney AI Centre and the School of Computer Science, in the Faculty of Engineering, The University of Sydney, Darlington, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919880273"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3200245"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-017-3241-z"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2972281"},{"key":"ref5","article-title":"Residual force control for agile human behavior imitation and extended motion synthesis","author":"Yuan","year":"2020"},{"key":"ref6","article-title":"Concrete problems in AI safety","author":"Amodei","year":"2016"},{"key":"ref7","article-title":"Benchmarking safe exploration in deep reinforcement learning","author":"Ray","year":"2019"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103500"},{"key":"ref9","article-title":"Neural bridge sampling for evaluating safety-critical autonomous systems","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Sinha"},{"key":"ref10","article-title":"Query-efficient imitation learning for end-to-end autonomous driving","author":"Zhang","year":"2016"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8968287"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793750"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/tits.2022.3227738"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2873794"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVE45908.2019.8965092"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3163747"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.metabol.2017.12.010"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.3141\/1830-09"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.2196\/11510"},{"key":"ref20","article-title":"Safe model-based reinforcement learning with robust cross-entropy method","author":"Liu","year":"2020"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.114818"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.trf.2008.03.002"},{"key":"ref23","article-title":"Learning from maps: Visual common sense for autonomous driving","author":"Seff","year":"2016"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2022.3152724"},{"key":"ref25","article-title":"Constrained policy optimization","author":"Achiam","year":"2017"},{"key":"ref26","article-title":"Responsive safety in reinforcement learning by PID Lagrangian methods","author":"Stooke","year":"2020"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20737"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"ref29","first-page":"4565","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/614"},{"issue":"1","key":"ref31","first-page":"1437","article-title":"A comprehensive survey on safe reinforcement learning","volume":"16","author":"Garc\u0131a","year":"2015","journal-title":"J. Mach. Learn. Res."},{"key":"ref32","article-title":"Projection-based constrained policy optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yang"},{"key":"ref33","first-page":"15338","article-title":"First order constrained optimization in policy space","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref34","article-title":"Constrained update projection approach to safe policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441"},{"key":"ref36","first-page":"13859","article-title":"Safe reinforcement learning by imagining the near future","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Thomas"},{"key":"ref37","article-title":"Constrained policy optimization via Bayesian world models","author":"As","year":"2022"},{"key":"ref38","first-page":"8092","article-title":"A Lyapunov-based approach to safe reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chow"},{"key":"ref39","article-title":"Lyapunov-based safe policy optimization for continuous control","author":"Chow","year":"2019"},{"key":"ref40","article-title":"Lyapunov-based uncertainty-aware safe reinforcement learning","author":"Jeddi","year":"2021"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1093\/oso\/9780198538677.003.0006"},{"key":"ref42","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Ross"},{"key":"ref43","first-page":"1","article-title":"Apprenticeship learning via inverse reinforcement learning","volume-title":"Proc. 21st Int. Conf. Mach. Learn.","author":"Abbeel"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3096966"},{"key":"ref45","article-title":"Generative adversarial imitation from observation","author":"Torabi","year":"2018"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3204708"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3054912"},{"key":"ref48","article-title":"Imitation learning from imperfect demonstration","author":"Wu","year":"2019"},{"key":"ref49","first-page":"1117","article-title":"When will generative adversarial imitation learning algorithms attain global convergence","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Guan"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1604.01685"},{"key":"ref51","first-page":"9407","article-title":"Variational imitation learning with diverse-quality demonstrations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tangkaratt"},{"issue":"1","key":"ref52","first-page":"48","article-title":"Even experts make mistakes","volume":"39","author":"Best","year":"1992","journal-title":"Risk Manage."},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.3354\/meps247017"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2019.8852307"},{"key":"ref55","article-title":"Towards practical adam: Non-convexity, convergence theory, and mini-batch acceleration","author":"Chen","year":"2021"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6021"},{"key":"ref57","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015"},{"key":"ref58","first-page":"2089","article-title":"Learning Bregman distance functions and its application for semi-supervised clustering","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1016\/0041-5553(67)90040-7"},{"key":"ref60","first-page":"49","article-title":"On the generalised distance in statistics","volume-title":"Proc. Nat. Inst. Sci. India","author":"Chandra"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/18.61115"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729694"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2020.2024"},{"key":"ref64","first-page":"12163","article-title":"Leverage the average: An analysis of KL regularization in reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Vieillard"},{"issue":"1","key":"ref65","first-page":"4431","article-title":"On the theory of policy gradient methods: Optimality, approximation, and distribution shift","volume":"22","author":"Agarwal","year":"2021","journal-title":"J. Mach. Learn. Res."},{"key":"ref66","first-page":"20423","article-title":"Saut\u00e9 RL: Almost surely safe reinforcement learning using state augmentation","volume-title":"Proc. Int. Conf. on Mach. Learn.","author":"Sootla"},{"key":"ref67","article-title":"Openai baselines","author":"Dhariwal","year":"2017"},{"key":"ref68","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"ref70","first-page":"24432","article-title":"Model-based safe deep reinforcement learning via a constrained proximal policy optimization algorithm","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jayant"},{"key":"ref71","article-title":"Safe reinforcement learning from pixels using a stochastic latent representation","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Hogewind"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17272"},{"key":"ref73","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Sutton"},{"key":"ref74","first-page":"4358","article-title":"Improving sample complexity bounds for (natural) actor-critic algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10241246\/10158032.pdf?arnumber=10158032","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,6]],"date-time":"2024-06-06T17:05:45Z","timestamp":1717693545000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10158032\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10]]},"references-count":74,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2023.3287908","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10]]}}}