{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,24]],"date-time":"2025-08-24T00:01:08Z","timestamp":1755993668628,"version":"3.44.0"},"reference-count":23,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T00:00:00Z","timestamp":1751932800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,7,8]],"date-time":"2025-07-08T00:00:00Z","timestamp":1751932800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,7,8]]},"DOI":"10.23919\/acc63710.2025.11107904","type":"proceedings-article","created":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T18:17:51Z","timestamp":1755800271000},"page":"3551-3557","source":"Crossref","is-referenced-by-count":0,"title":["Online Reinforcement Learning with Passive Memory"],"prefix":"10.23919","author":[{"given":"Anay","family":"Pattanaik","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign,Department of Computer Science and Coordinated Science Laboratory,Urbana,IL,USA,61801"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lav R.","family":"Varshney","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign,Department of Electrical and Computer Engineering and Coordinated Science Laboratory,Urbana,IL,USA,61801"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"volume-title":"Reinforcement Learning: An Introduction.","year":"2018","author":"Sutton","key":"ref1"},{"key":"ref2","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Silver"},{"key":"ref3","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Schulman"},{"year":"2017","author":"Schulman","article-title":"Proximal policy optimization algorithms","key":"ref4"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1038\/nature14236"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1609\/aaai.v30i1.10295"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.32657\/10356\/90191"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1016\/B978-0-12-375000-6.00152-X"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1007\/978-3-540-45702-2_8"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1146\/annurev-psych-122414-033625"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1093\/nc\/niad020"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1007\/s10539-020-09772-0"},{"year":"2020","author":"Levine","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","key":"ref13"},{"key":"ref14","first-page":"378","article-title":"Offline reinforcement learning under value and density-ratio realizability: the power of gaps","volume-title":"Proceedings of the Conference on Uncertainty in Artificial Intelligence","author":"Chen"},{"volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"Song","article-title":"Hybrid RL: Using both offline and online data can make RL efficient","key":"ref15"},{"key":"ref16","first-page":"1577","article-title":"Efficient online reinforcement learning with offline data","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Ball"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1109\/ijcnn.2003.1223891"},{"year":"2019","author":"Nachum","article-title":"AlgaeDICE: Policy gradient from arbitrary experience","key":"ref18"},{"volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming.","year":"2014","author":"Puterman","key":"ref19"},{"year":"2020","author":"Nachum","article-title":"Reinforcement learning via Fenchel-Rockafellar duality","key":"ref20"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1007\/b13794"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1093\/acprof:oso\/9780199535255.001.0001"},{"volume-title":"The Theory of Probabilities.","year":"1946","author":"Bernstein","key":"ref23"}],"event":{"name":"2025 American Control Conference (ACC)","start":{"date-parts":[[2025,7,8]]},"location":"Denver, CO, USA","end":{"date-parts":[[2025,7,10]]}},"container-title":["2025 American Control Conference (ACC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11107441\/11107442\/11107904.pdf?arnumber=11107904","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T23:57:39Z","timestamp":1755907059000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11107904\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,8]]},"references-count":23,"URL":"https:\/\/doi.org\/10.23919\/acc63710.2025.11107904","relation":{},"subject":[],"published":{"date-parts":[[2025,7,8]]}}}