{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,9]],"date-time":"2026-08-09T21:18:40Z","timestamp":1786310320228,"version":"3.56.0"},"reference-count":45,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"7","license":[{"start":{"date-parts":[[2022,7,1]],"date-time":"2022-07-01T00:00:00Z","timestamp":1656633600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,7,1]],"date-time":"2022-07-01T00:00:00Z","timestamp":1656633600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,7,1]],"date-time":"2022-07-01T00:00:00Z","timestamp":1656633600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976215"],"award-info":[{"award-number":["61976215"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772532"],"award-info":[{"award-number":["61772532"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1813203"],"award-info":[{"award-number":["U1813203"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Syst. Man Cybern, Syst."],"published-print":{"date-parts":[[2022,7]]},"DOI":"10.1109\/tsmc.2021.3098451","type":"journal-article","created":{"date-parts":[[2021,7,29]],"date-time":"2021-07-29T19:58:19Z","timestamp":1627588699000},"page":"4600-4610","source":"Crossref","is-referenced-by-count":218,"title":["Proximal Policy Optimization With Policy Feedback"],"prefix":"10.1109","volume":"52","author":[{"given":"Yang","family":"Gu","sequence":"first","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2022-9999","authenticated-orcid":false,"given":"Yuhu","family":"Cheng","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5451-7230","authenticated-orcid":false,"given":"C. L. Philip","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Computer Science and Engineering, South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5327-1088","authenticated-orcid":false,"given":"Xuesong","family":"Wang","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education and the School of Information and Control Engineering, China University of Mining and Technology, Xuzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Playing Atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2019.2957051"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2019.2963246"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2929141"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2020.3012832"},{"key":"ref6","volume-title":"Deep reinforcement learning for autonomous driving","author":"Wang","year":"2018"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2018.2870983"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref9","volume-title":"Prioritized experience replay","author":"Schaul","year":"2015"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2927227"},{"key":"ref12","volume-title":"Noisy networks for exploration","author":"Fortunato","year":"2017"},{"key":"ref13","volume-title":"Distributed deep Q-Learning","author":"Ong","year":"2015"},{"key":"ref14","volume-title":"Dueling network architectures for deep reinforcement learning","author":"Wang","year":"2015"},{"key":"ref15","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2006.282564"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-30164-8"},{"key":"ref18","first-page":"605","article-title":"Deterministic policy gradient algorithms","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Silver"},{"key":"ref19","volume-title":"Neural policy gradient methods: Global optimality and rates of convergence","author":"Wang","year":"2019"},{"key":"ref20","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref21","volume-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11607"},{"key":"ref23","first-page":"8092","article-title":"A Lyapunov-based approach to safe reinforcement learning","volume-title":"Proc. Conf. Neural Inf. Process. Syst.","author":"Chow"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2017.2712188"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2019.2926167"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref27","volume-title":"Trust-PCL: An off-policy trust region method for continuous control","author":"Nachum","year":"2017"},{"key":"ref28","first-page":"1578","article-title":"Path consistency learning in tsallis entropy regularized MDPs","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Nachum"},{"key":"ref29","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref30","volume-title":"Equivalence between policy gradients and soft Q-learning","author":"Schulman","year":"2017"},{"key":"ref31","first-page":"2776","article-title":"Bridging the gap between value and policy based reinforcement learning","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","author":"Nachum"},{"key":"ref32","volume-title":"Soft actor\u2013critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018"},{"key":"ref33","volume-title":"Reinforcement learning through asynchronous advantage actor\u2013critic on a GPU","author":"Babaeizadeh","year":"2016"},{"key":"ref34","volume-title":"Are deep policy gradient algorithms truly policy gradient algorithms?","author":"Ilyas","year":"2018"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2007.09.009"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-16952-6_49"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICORR.2011.5975338"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CCIS.2018.8691217"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2012.6343862"},{"key":"ref40","volume-title":"How to discount deep reinforcement learning: Towards new dynamic strategies","author":"Francois-Lavet","year":"2015"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/DevLrn.2013.6652533"},{"key":"ref42","first-page":"1","article-title":"Variance reduction techniques for gradient estimates in reinforcement learning","volume-title":"Proc. Annu. Neural Inf. Process. Syst. Conf.","author":"Greensmith"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ADPRL.2009.4927542"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11607"},{"key":"ref45","first-page":"267","article-title":"Approximately optimal approximate reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kakade"}],"container-title":["IEEE Transactions on Systems, Man, and Cybernetics: Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6221021\/9797000\/09501950.pdf?arnumber=9501950","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,11]],"date-time":"2024-01-11T23:38:11Z","timestamp":1705016291000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9501950\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7]]},"references-count":45,"journal-issue":{"issue":"7"},"URL":"https:\/\/doi.org\/10.1109\/tsmc.2021.3098451","relation":{},"ISSN":["2168-2216","2168-2232"],"issn-type":[{"value":"2168-2216","type":"print"},{"value":"2168-2232","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,7]]}}}