{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T09:47:20Z","timestamp":1784454440996,"version":"3.55.0"},"reference-count":44,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176259"],"award-info":[{"award-number":["62176259"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976215"],"award-info":[{"award-number":["61976215"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013058","name":"Key Research and Development Program of Jiangsu Province","doi-asserted-by":"publisher","award":["BE2022095"],"award-info":[{"award-number":["BE2022095"]}],"id":[{"id":"10.13039\/501100013058","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Intell. Transport. Syst."],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1109\/tits.2023.3292253","type":"journal-article","created":{"date-parts":[[2023,7,17]],"date-time":"2023-07-17T17:53:10Z","timestamp":1689616390000},"page":"14320-14328","source":"Crossref","is-referenced-by-count":29,"title":["Autonomous Driving Based on Approximate Safe Action"],"prefix":"10.1109","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5327-1088","authenticated-orcid":false,"given":"Xuesong","family":"Wang","sequence":"first","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, the Xuzhou Key Laboratory of Artificial Intelligence and Big Data, and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8963-6374","authenticated-orcid":false,"given":"Jiazhi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, the Xuzhou Key Laboratory of Artificial Intelligence and Big Data, and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9427-1532","authenticated-orcid":false,"given":"Diyuan","family":"Hou","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, the Xuzhou Key Laboratory of Artificial Intelligence and Big Data, and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2022-9999","authenticated-orcid":false,"given":"Yuhu","family":"Cheng","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, the Xuzhou Key Laboratory of Artificial Intelligence and Big Data, and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"10","key":"ref1","first-page":"3884","article-title":"A reinforcement learning approach to autonomous decision making of intelligent vehicles on highways","volume":"50","author":"Xu","year":"2020","journal-title":"IEEE Trans. Syst., Man, Cybern. Syst."},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2020.3047129"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2961739"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICAIS50930.2021.9395812"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3105905"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref8","first-page":"13644","article-title":"Constrained variational policy optimization for safe reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liu"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2020.1003474"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2021.1004395"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2020.3046643"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2919865"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2021.1004353"},{"key":"ref14","article-title":"Reward constrained policy optimization","author":"Tessler","year":"2018","journal-title":"arXiv:1805.11074"},{"key":"ref15","first-page":"11795","article-title":"Accelerating safe reinforcement learning with constraint-mismatched baseline policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Yang"},{"key":"ref16","volume-title":"Constrained Markov Decision Processes: Stochastic Modeling","author":"Altman","year":"1999"},{"key":"ref17","first-page":"8378","article-title":"Natural policy gradient primal-dual method for constrained Markov decision processes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ding"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3238656"},{"key":"ref19","article-title":"SafeRL-kit: Evaluating efficient reinforcement learning methods for safe autonomous driving","author":"Zhang","year":"2022","journal-title":"arXiv:2206.08528"},{"key":"ref20","first-page":"9133","article-title":"Responsive safety in reinforcement learning by PID Lagrangian methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Stooke"},{"issue":"1","key":"ref21","first-page":"6070","article-title":"Risk-constrained reinforcement learning with percentile risk criteria","volume":"18","author":"Chow","year":"2017","journal-title":"J. Mach. Learn. Res."},{"key":"ref22","article-title":"Feasible actor-critic: Constrained reinforcement learning for ensuring statewise safety","author":"Ma","year":"2021","journal-title":"arXiv:2105.10682"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5932"},{"key":"ref24","first-page":"8502","article-title":"Constrained Markov decision processes via backward value functions","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Satija"},{"key":"ref25","first-page":"1554","article-title":"Safe driving via expert guided policy optimization","volume-title":"Proc. Conf. Robot Learn.","author":"Peng"},{"key":"ref26","article-title":"Safe exploration in continuous action spaces","author":"Dalal","year":"2018","journal-title":"arXiv:1801.08757"},{"key":"ref27","first-page":"908","article-title":"Safe model-based reinforcement learning with stability guarantees","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Berkenkamp"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3070252"},{"key":"ref29","first-page":"25621","article-title":"Learning barrier certificates: Towards safe reinforcement learning with zero training-time violations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Luo"},{"key":"ref30","first-page":"1226","article-title":"Alwayssafe: Reinforcement learning without safety constraint violations during training","volume-title":"Proc. Int. Conf. Auto. Agents MultiAgent Syst.","author":"Sim\u00e3o"},{"key":"ref31","first-page":"8103","article-title":"A Lyapunov-based approach to safe reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chow"},{"key":"ref32","article-title":"Lyapunov barrier policy optimization","author":"Sikchi","year":"2021","journal-title":"arXiv:2103.09230"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"ref34","article-title":"Learning to be safe: Deep RL with a safety critic","author":"Srinivasan","year":"2020","journal-title":"arXiv:2010.14603"},{"key":"ref35","first-page":"10630","article-title":"Safe reinforcement learning using advantage-based intervention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wagener"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17272"},{"key":"ref37","first-page":"22","article-title":"Constrained policy optimization","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Achiam"},{"key":"ref38","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref39","article-title":"Projection-based constrained policy optimization","author":"Yang","year":"2020","journal-title":"arXiv:2010.03152"},{"key":"ref40","first-page":"15338","article-title":"First order constrained optimization in policy space","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref41","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref42","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"issue":"3","key":"ref43","first-page":"3461","article-title":"MetaDrive: Composing diverse driving scenarios for generalizable reinforcement learning","volume":"45","author":"Li","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref44","article-title":"Bullet-safety-gym: A framework for constrained reinforcement learning","author":"Gronauer","year":"2022"}],"container-title":["IEEE Transactions on Intelligent Transportation Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6979\/10339106\/10184325.pdf?arnumber=10184325","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,12]],"date-time":"2024-02-12T19:56:18Z","timestamp":1707767778000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10184325\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12]]},"references-count":44,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tits.2023.3292253","relation":{},"ISSN":["1524-9050","1558-0016"],"issn-type":[{"value":"1524-9050","type":"print"},{"value":"1558-0016","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12]]}}}