{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T16:23:53Z","timestamp":1783873433921,"version":"3.55.0"},"reference-count":44,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T00:00:00Z","timestamp":1706745600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T00:00:00Z","timestamp":1706745600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T00:00:00Z","timestamp":1706745600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,2]]},"DOI":"10.1109\/tpami.2023.3328397","type":"journal-article","created":{"date-parts":[[2023,10,30]],"date-time":"2023-10-30T19:01:13Z","timestamp":1698692473000},"page":"1199-1211","source":"Crossref","is-referenced-by-count":10,"title":["False Correlation Reduction for Offline Reinforcement Learning"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6088-7534","authenticated-orcid":false,"given":"Zhihong","family":"Deng","sequence":"first","affiliation":[{"name":"Faculty of Engineering and Information Technology, Australian Artificial Intelligence Institute, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9217-543X","authenticated-orcid":false,"given":"Zuyue","family":"Fu","sequence":"additional","affiliation":[{"name":"Department of Industrial Engineering and Management Sciences, Northwestern University, Evanston, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1654-4681","authenticated-orcid":false,"given":"Lingxiao","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Industrial Engineering and Management Sciences, Northwestern University, Evanston, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhuoran","family":"Yang","sequence":"additional","affiliation":[{"name":"Department of Statistics and Data Science, Yale University, New Haven, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8379-9385","authenticated-orcid":false,"given":"Chenjia","family":"Bai","sequence":"additional","affiliation":[{"name":"Shanghai Artificial Intelligence Laboratory, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5348-0632","authenticated-orcid":false,"given":"Tianyi","family":"Zhou","sequence":"additional","affiliation":[{"name":"Department of Computer Science and UMIACS, University of Maryland, College Park, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoran","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Industrial Engineering and Management Sciences, Northwestern University, Evanston, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5301-7779","authenticated-orcid":false,"given":"Jing","family":"Jiang","sequence":"additional","affiliation":[{"name":"Faculty of Engineering and Information Technology, Australian Artificial Intelligence Institute, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref2","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020"},{"key":"ref3","article-title":"Behavior regularized offline reinforcement learning","author":"Wu","year":"2019"},{"key":"ref4","first-page":"11784","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kumar"},{"key":"ref5","first-page":"5084","article-title":"Is pessimism provably efficient for offline RL?","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Jin"},{"key":"ref6","first-page":"6683","article-title":"Bellman-consistent pessimism for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xie"},{"key":"ref7","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020"},{"key":"ref8","article-title":"Pessimistic bootstrapping for uncertainty-driven offline reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bai"},{"key":"ref9","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kumar"},{"key":"ref10","first-page":"14 129","article-title":"MOPO: Model-based offline policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu"},{"key":"ref11","first-page":"4033","article-title":"Deep exploration via bootstrapped DQN","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Osband"},{"key":"ref12","first-page":"6405","article-title":"Simple and scalable predictive uncertainty estimation using deep ensembles","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lakshminarayanan"},{"key":"ref13","first-page":"8626","article-title":"Randomized prior functions for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Osband"},{"key":"ref14","first-page":"1787","article-title":"Better exploration with optimistic actor critic","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ciosek"},{"key":"ref15","first-page":"6131","article-title":"SUNRISE: A simple unified framework for ensemble learning in deep reinforcement learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Lee"},{"key":"ref16","first-page":"28 954","article-title":"COMBO: Conservative offline model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu"},{"key":"ref17","article-title":"Batch reinforcement learning through continuation method","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Guo"},{"key":"ref18","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"Proc. 31st Int. Conf. Mach. Learn.","author":"Silver"},{"key":"ref19","article-title":"Benchmarking batch deep reinforcement learning algorithms","author":"Fujimoto","year":"2019"},{"key":"ref20","first-page":"3682","article-title":"EMaQ: Expected-max Q-learning operator for simple yet effective offline and online RL","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Ghasemipour"},{"key":"ref21","article-title":"AWAC: Accelerating online reinforcement learning with offline datasets","author":"Nair","year":"2021"},{"key":"ref22","first-page":"5774","article-title":"Offline reinforcement learning with fisher divergence critic regularization","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Kostrikov"},{"key":"ref23","first-page":"11 319","article-title":"Uncertainty weighted actor-critic for offline reinforcement learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Wu"},{"key":"ref24","first-page":"20 132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Fujimoto"},{"key":"ref25","first-page":"21 810","article-title":"MOReL: Model-based offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kidambi"},{"key":"ref26","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified Q-ensemble","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"An"},{"key":"ref27","first-page":"1447","article-title":"More robust doubly robust off-policy evaluation","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Farajtabar"},{"key":"ref28","first-page":"5361","article-title":"Breaking the curse of horizon: Infinite-horizon off-policy estimation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liu"},{"key":"ref29","first-page":"9668","article-title":"Towards optimal off-policy evaluation for reinforcement learning with marginalized importance sampling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xie"},{"key":"ref30","first-page":"2318","article-title":"DualDICE: Behavior-agnostic estimation of discounted stationary distribution corrections","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Nachum"},{"key":"ref31","first-page":"2747","article-title":"Minimax value interval for off-policy evaluation and policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jiang"},{"key":"ref32","first-page":"2701","article-title":"Minimax-optimal off-policy evaluation with linear function approximation","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Duan"},{"key":"ref33","first-page":"6551","article-title":"Off-policy evaluation via the regularized Lagrangian","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref34","first-page":"1567","article-title":"Near-optimal provable uniform convergence in offline policy evaluation for reinforcement learning","volume-title":"Proc. 24th Int. Conf. Artif. Intell. Statist.","author":"Yin"},{"key":"ref35","article-title":"GenDICE: Generalized offline estimation of stationary values","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang"},{"issue":"49","key":"ref36","first-page":"1629","article-title":"Approximate modified policy iteration and its application to the game of tetris","volume":"16","author":"Scherrer","year":"2015","journal-title":"J. Mach. Learn. Res."},{"key":"ref37","first-page":"1042","article-title":"Information-theoretic considerations in batch reinforcement learning","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Chen"},{"key":"ref38","first-page":"550","article-title":"Q* approximation schemes for batch reinforcement learning: A theoretical comparison","volume-title":"Proc. 36th Conf. Uncertainty Artif. Intell.","author":"Xie"},{"key":"ref39","first-page":"486","article-title":"A theoretical analysis of deep Q-learning","volume-title":"Proc. 2nd Conf. Learn. Dyn. Control","author":"Fan"},{"key":"ref40","first-page":"11 404","article-title":"Batch value-function approximation with only realizability","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Xie"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1214\/22-AOS2231"},{"key":"ref42","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref43","first-page":"29 304","article-title":"Deep reinforcement learning at the edge of the statistical precipice","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Agarwal"},{"key":"ref44","first-page":"1283","article-title":"Provably efficient exploration in policy optimization","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Cai"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10384454\/10301548.pdf?arnumber=10301548","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T18:55:36Z","timestamp":1705085736000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10301548\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2]]},"references-count":44,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2023.3328397","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2]]}}}