{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T21:31:59Z","timestamp":1781818319118,"version":"3.54.5"},"reference-count":48,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"NSF","award":["ECCS-1931600"],"award-info":[{"award-number":["ECCS-1931600"]}]},{"name":"NSF","award":["DMS-1664644"],"award-info":[{"award-number":["DMS-1664644"]}]},{"name":"NSF","award":["CNS-1645681"],"award-info":[{"award-number":["CNS-1645681"]}]},{"name":"NSF","award":["IIS-1914792"],"award-info":[{"award-number":["IIS-1914792"]}]},{"name":"NSF","award":["DEB-2433726"],"award-info":[{"award-number":["DEB-2433726"]}]},{"name":"NSF","award":["ECCS-2317079"],"award-info":[{"award-number":["ECCS-2317079"]}]},{"name":"NSF","award":["CCF-2200052"],"award-info":[{"award-number":["CCF-2200052"]}]},{"DOI":"10.13039\/100000006","name":"ONR","doi-asserted-by":"publisher","award":["N00014-19-1-2571"],"award-info":[{"award-number":["N00014-19-1-2571"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"name":"NIH","award":["R01 GM135930"],"award-info":[{"award-number":["R01 GM135930"]}]},{"name":"NIH","award":["UL54 TR004130"],"award-info":[{"award-number":["UL54 TR004130"]}]},{"name":"DOE","award":["DE-AR0001282"],"award-info":[{"award-number":["DE-AR0001282"]}]},{"name":"DOE","award":["DE-EE0009696"],"award-info":[{"award-number":["DE-EE0009696"]}]},{"DOI":"10.13039\/100000181","name":"Air Force Office of Scientific Research","doi-asserted-by":"publisher","award":["FA9550-19-1-0158"],"award-info":[{"award-number":["FA9550-19-1-0158"]}],"id":[{"id":"10.13039\/100000181","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014600","name":"MathWorks","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014600","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007161","name":"Boston University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100007161","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100014462","name":"Mitsubishi Electric Research Laboratories","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014462","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1109\/tac.2024.3454011","type":"journal-article","created":{"date-parts":[[2024,9,3]],"date-time":"2024-09-03T17:45:39Z","timestamp":1725385539000},"page":"1236-1243","source":"Crossref","is-referenced-by-count":6,"title":["Generalized Policy Improvement Algorithms With Theoretically Supported Sample Reuse"],"prefix":"10.1109","volume":"70","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0655-3637","authenticated-orcid":false,"given":"James","family":"Queeney","sequence":"first","affiliation":[{"name":"Mitsubishi Electric Research Laboratories, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3343-2913","authenticated-orcid":false,"given":"Ioannis Ch.","family":"Paschalidis","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering and Division of Systems Engineering, Boston University, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1625-7658","authenticated-orcid":false,"given":"Christos G.","family":"Cassandras","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering and Division of Systems Engineering, Boston University, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.arcontrol.2018.09.005"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-053018-023825"},{"key":"ref3","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","volume-title":"Proc. 33 rd Int. Conf. Mach. Learn.","volume":"48","author":"Duan","year":"2016"},{"key":"ref4","article-title":"Model-ensemble trust-region policy optimization","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Kurutach","year":"2018"},{"key":"ref5","article-title":"When to trust your model: Model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Janner","year":"2019"},{"key":"ref6","first-page":"7953","article-title":"A game theoretic framework for model based reinforcement learning","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Rajeswaran","year":"2020"},{"key":"ref7","article-title":"Behavior regularized offline reinforcement learning","author":"Wu","year":"2019"},{"key":"ref8","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kumar","year":"2020"},{"key":"ref9","first-page":"21810","article-title":"MOReL: Model-based offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kidambi","year":"2020"},{"key":"ref10","first-page":"22","article-title":"Constrained policy optimization","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Achiam","year":"2017"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2018.2876389"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3049335"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-042920-020211"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2022.3152724"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160293"},{"key":"ref16","first-page":"11909","article-title":"Generalized proximal policy optimization with sample reuse","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Queeney","year":"2021"},{"key":"ref17","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref18","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","volume":"37","author":"Schulman","year":"2015"},{"key":"ref19","article-title":"V-MPO: On-policy maximum a posteriori policy optimization for discrete and continuous control","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Song","year":"2020"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.simpa.2020.100022"},{"key":"ref21","first-page":"267","article-title":"Approximately optimal approximate reinforcement learning","volume-title":"Proc. 19th Int. Conf. Mach. Learn","author":"Kakade","year":"2002"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"ref23","article-title":"Implementation matters in deep RL: A case study on PPO and TRPO","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Engstrom","year":"2020"},{"key":"ref24","article-title":"What matters for on-policy deep actor-critic methods? A large-scale study","volume-title":"Proc. 9th Int. Conf. Learn. Representations","author":"Andrychowicz","year":"2021"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i11.17130"},{"key":"ref26","article-title":"Trust region-guided proximal policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Wang","year":"2019"},{"key":"ref27","first-page":"113","article-title":"Truly proximal policy optimization","volume-title":"Proc. 35th Uncertainty Artif. Intell. Conf.","volume":"115","author":"Wang","year":"2020"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3051456"},{"key":"ref29","article-title":"Supervised policy update for deep reinforcement learning","volume-title":"Proc. 7th Int. Conf. Learn. Representations","author":"Vuong","year":"2019"},{"key":"ref30","article-title":"Continuous control with deep reinforcement learning","volume-title":"Proc. 4th Int. Conf. Learn. Representations","author":"Lillicrap","year":"2016"},{"key":"ref31","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Fujimoto","year":"2018"},{"key":"ref32","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Haarnoja","year":"2018"},{"key":"ref33","article-title":"Maximum a posteriori policy optimisation","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Abdolmaleki","year":"2018"},{"key":"ref34","article-title":"Combining policy gradient and Q-learning","volume-title":"Proc. 5th Int. Conf. Learn. Representations","author":"O\u2019Donoghue","year":"2017"},{"key":"ref35","article-title":"Q-prop: Sample-efficient policy gradient with an off-policy critic","volume-title":"Proc. 5th Int. Conf. Learn. Representations","author":"Gu","year":"2017"},{"key":"ref36","article-title":"Interpolated policy gradient: Merging on-policy and off-policy gradient estimation for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Gu","year":"2017"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.3044196"},{"key":"ref38","first-page":"1017","article-title":"P3O: Policy-on policy-off policy optimization","volume-title":"Proc. 35th Uncertainty Artif. Intell. Conf.","volume":"115","author":"Fakoor","year":"2020"},{"key":"ref39","article-title":"Sample efficient actor-critic with experience replay","volume-title":"Proc. 5th Int. Conf. Learn. Representations","author":"Wang","year":"2017"},{"key":"ref40","first-page":"4851","article-title":"Remember and forget for experience replay","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","volume":"97","author":"Novati","year":"2019"},{"key":"ref41","first-page":"10070","article-title":"Striving for simplicity and performance in off-policy DRL: Output normalization and non-uniform sampling","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Wang","year":"2020"},{"key":"ref42","article-title":"Prioritized experience replay","volume-title":"Proc. 4th Int. Conf. Learn. Representations","author":"Schaul","year":"2016"},{"issue":"9","key":"ref43","first-page":"1","article-title":"Experience selection in deep reinforcement learning for control","volume":"19","author":"Bruin","year":"2018","journal-title":"J. Mach. Learn. Res."},{"key":"ref44","article-title":"A note on importance sampling using standardized weights","author":"Kong","year":"1992"},{"key":"ref45","article-title":"Diversity-driven exploration strategy for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Hong","year":"2018"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/b13794"},{"key":"ref47","article-title":"High-dimensional continuous control using generalized advantage estimation","volume-title":"Proc. 4th Int. Conf. Learn. Representations","author":"Schulman","year":"2016"},{"key":"ref48","first-page":"1407","article-title":"IMPALA: Scalable distributed deep-RL with importance weighted actor-learner architectures","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","volume":"80","author":"Espeholt","year":"2018"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/9\/10857662\/10663867-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9\/10857662\/10663867.pdf?arnumber=10663867","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,30]],"date-time":"2025-01-30T19:28:10Z","timestamp":1738265290000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10663867\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2]]},"references-count":48,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tac.2024.3454011","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"value":"0018-9286","type":"print"},{"value":"1558-2523","type":"electronic"},{"value":"2334-3303","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2]]}}}