{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T05:47:22Z","timestamp":1783748842264,"version":"3.55.0"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1109\/icra40945.2020.9196867","type":"proceedings-article","created":{"date-parts":[[2020,9,15]],"date-time":"2020-09-15T21:25:46Z","timestamp":1600205146000},"page":"7166-7172","source":"Crossref","is-referenced-by-count":47,"title":["Robust Model Predictive Shielding for Safe Reinforcement Learning with Stochastic Dynamics"],"prefix":"10.1109","author":[{"given":"Shuo","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Osbert","family":"Bastani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","first-page":"-387i","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"0"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.014"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2014.6942859"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2011.6160389"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1080\/00423114.2014.902537"},{"key":"ref30","article-title":"Mamps: Safe multi-agent reinforcement learning via model predictive shielding","author":"zhang","year":"0"},{"key":"ref37","author":"vapnik","year":"2013","journal-title":"The Nature of Statistical Learning Theory"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/j.arcontrol.2016.04.006"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1002\/rnc.1758"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1177\/0278364916647192"},{"key":"ref10","article-title":"Learning safe unlabeled multirobot planning with motion constraints","author":"khan","year":"2019","journal-title":"CoRR"},{"key":"ref11","article-title":"Graph policy gradients for large scale robot control","author":"khan","year":"2019","journal-title":"CoRL"},{"key":"ref12","article-title":"Learning decentralized controllers for robot swarms with graph neural networks","author":"tolstaya","year":"2019","journal-title":"CoRL"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2015.2505910"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/0005-1098(89)90002-2"},{"key":"ref15","author":"rawlings","year":"2009","journal-title":"Model Predictive Control Theory and Design"},{"key":"ref16","article-title":"Generative adversarial imitation learning","author":"ho","year":"2016","journal-title":"CoRR"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2012.6225136"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2014.7039601"},{"key":"ref19","first-page":"908","article-title":"Safe model-based reinforcement learning with stability guarantees","author":"berkenkamp","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref28","article-title":"Safe planning via model predictive shielding","author":"bastani","year":"2019","journal-title":"CoRR"},{"key":"ref4","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"CoRR"},{"key":"ref27","article-title":"Reachable set estimation and verification for neural network models of nonlinear dynamic systems","author":"xiang","year":"2018","journal-title":"CoRR"},{"key":"ref3","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017","journal-title":"CoRR"},{"key":"ref6","article-title":"Hierarchical policy design for sample-efficient learning of robot table tennis through self-play","author":"mahjourian","year":"2018","journal-title":"CoRR"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619829"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref8","article-title":"End to end learning for selfdriving cars","author":"bojarski","year":"2016","journal-title":"CoRR"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2930489"},{"key":"ref2","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"CoRR"},{"key":"ref9","article-title":"Learning dexterous in-hand manipulation","author":"andrychowicz","year":"2018"},{"key":"ref1","article-title":"Deep reinforcement learning with double q-learning","author":"van hasselt","year":"2015","journal-title":"CoRR"},{"key":"ref20","first-page":"2494","article-title":"Verifiable reinforcement learning via policy extraction","author":"bastani","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794107"},{"key":"ref21","doi-asserted-by":"crossref","DOI":"10.1609\/aaai.v32i1.11797","article-title":"Safe reinforcement learning via shielding","author":"alshiekh","year":"2018","journal-title":"Thirty-Second AAAI Conference on Artificial Intelligence"},{"key":"ref24","article-title":"Safety verification and robustness analysis of neural networks via quadratic constraints and semidefinite programming","author":"fazlyab","year":"2019","journal-title":"CoRR"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3302504.3311806"},{"key":"ref26","article-title":"Safety verification of cyber-physical systems with reinforcement learning control, emsoft 2019","author":"tran","year":"2019"},{"key":"ref25","article-title":"Efficient and accurate estimation of lipschitz constants for deep neural networks","author":"fazlyab","year":"2019","journal-title":"CoRR"}],"event":{"name":"2020 IEEE International Conference on Robotics and Automation (ICRA)","location":"Paris, France","start":{"date-parts":[[2020,5,31]]},"end":{"date-parts":[[2020,8,31]]}},"container-title":["2020 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9187508\/9196508\/09196867.pdf?arnumber=9196867","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,18]],"date-time":"2022-11-18T13:57:09Z","timestamp":1668779829000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9196867\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/icra40945.2020.9196867","relation":{},"subject":[],"published":{"date-parts":[[2020,5]]}}}