{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T05:13:33Z","timestamp":1777958013638,"version":"3.51.4"},"reference-count":56,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2023YFF0905400"],"award-info":[{"award-number":["2023YFF0905400"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2341229"],"award-info":[{"award-number":["U2341229"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976102"],"award-info":[{"award-number":["61976102"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U19A2065"],"award-info":[{"award-number":["U19A2065"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476110"],"award-info":[{"award-number":["62476110"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Research and Development Project of Jilin Province","award":["20240304200SF"],"award-info":[{"award-number":["20240304200SF"]}]},{"name":"International Cooperation Project of Jilin Province","award":["20220402009GH"],"award-info":[{"award-number":["20220402009GH"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/tnnls.2025.3633997","type":"journal-article","created":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T18:29:17Z","timestamp":1764095357000},"page":"2456-2468","source":"Crossref","is-referenced-by-count":3,"title":["A Simple Unified Uncertainty-Guided Framework for Offline-to-Online Reinforcement Learning"],"prefix":"10.1109","volume":"37","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9304-5405","authenticated-orcid":false,"given":"Siyuan","family":"Guo","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence and the International Center of Future Science, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1137-9939","authenticated-orcid":false,"given":"Yanchao","family":"Sun","sequence":"additional","affiliation":[{"name":"JPMorgan AI Research, New York, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8658-9447","authenticated-orcid":false,"given":"Jifeng","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and the International Center of Future Science, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5387-7904","authenticated-orcid":false,"given":"Sili","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and the International Center of Future Science, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7835-9556","authenticated-orcid":false,"given":"Hechang","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and the International Center of Future Science, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8519-4750","authenticated-orcid":false,"given":"Haiyin","family":"Piao","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1539-7939","authenticated-orcid":false,"given":"Lichao","family":"Sun","sequence":"additional","affiliation":[{"name":"Lehigh University, Bethlehem, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2697-8093","authenticated-orcid":false,"given":"Yi","family":"Chang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence and the International Center of Future Science, Jilin University, Changchun, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv:2005.01643"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3250269"},{"key":"ref3","first-page":"2041","article-title":"Offline reinforcement learning at multiple frequencies","volume-title":"Proc. Conf. Robot Learn.","author":"Burns"},{"key":"ref4","article-title":"Clinical decision transformer: Intended treatment recommendation through goal prompting","author":"Lee","year":"2023","journal-title":"arXiv:2302.00612"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583418"},{"key":"ref6","first-page":"1702","article-title":"Offline-to-online reinforcement learning via balanced replay and pessimistic Q-ensemble","volume-title":"Proc. Conf. Robot Learn.","author":"Lee"},{"key":"ref7","first-page":"1","article-title":"Policy expansion for bridging offline-to-online reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhang"},{"key":"ref8","first-page":"27042","article-title":"Online decision transformer","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zheng"},{"key":"ref9","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref10","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kumar"},{"key":"ref11","first-page":"1","article-title":"Fine-tuning offline policies with optimistic action selection","volume-title":"Proc. Deep Reinforcement Learn. Workshop NeurIPS","author":"Mark"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i9.26345"},{"key":"ref13","first-page":"6131","article-title":"SUNRISE: A simple unified framework for ensemble learning in deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lee"},{"key":"ref14","first-page":"1","article-title":"Pessimistic bootstrapping for uncertainty-driven offline reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Bai"},{"key":"ref15","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified q-ensemble","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"An"},{"key":"ref16","first-page":"11319","article-title":"Uncertainty weighted actor-critic for offline reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wu"},{"key":"ref17","first-page":"1","article-title":"Auto-encoding variational Bayes","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"},{"key":"ref18","first-page":"213","article-title":"R-MAX\u2014A general polynomial time algorithm for near-optimal reinforcement learning","volume":"3","author":"Brafman","year":"2002","journal-title":"J. Mach. Learn. Res."},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.tcs.2009.01.016"},{"key":"ref20","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020","journal-title":"arXiv:2004.07219"},{"key":"ref21","article-title":"ENOTO: Improving offline-to-online reinforcement learning with Q-ensembles","author":"Zhao","year":"2023","journal-title":"arXiv:2306.06871"},{"key":"ref22","first-page":"1577","article-title":"Efficient online reinforcement learning with offline data","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ball"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3293508"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3443102"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3443082"},{"key":"ref26","first-page":"1","article-title":"Stabilizing off-policy q-learning via bootstrapping error reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Kumar"},{"key":"ref27","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Fujimoto"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2268"},{"key":"ref29","first-page":"1711","article-title":"Mildly conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Conf. Neural Inf. Process. Syst.","author":"Lyu"},{"key":"ref30","first-page":"5774","article-title":"Offline reinforcement learning with Fisher divergence critic regularization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kostrikov"},{"key":"ref31","article-title":"AWAC: Accelerating online reinforcement learning with offline datasets","author":"Nair","year":"2020","journal-title":"arXiv:2006.09359"},{"key":"ref32","first-page":"1","article-title":"Offline reinforcement learning with implicit Q-learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kostrikov"},{"key":"ref33","first-page":"40452","article-title":"Actor-critic alignment for offline-to-online reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"202","author":"Yu"},{"key":"ref34","article-title":"Cal-QL: Calibrated offline RL pre-training for efficient online fine-tuning","author":"Nakamoto","year":"2023","journal-title":"arXiv:2303.05479"},{"key":"ref35","article-title":"Finetuning from offline reinforcement learning: Challenges, trade-offs and practical solutions","author":"Luo","year":"2023","journal-title":"arXiv:2303.17396"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2023.3302804"},{"key":"ref37","first-page":"15084","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Chen"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.14428\/esann\/2022.ES2022-110"},{"key":"ref39","article-title":"PROTO: Iterative policy regularized offline-to-online reinforcement learning","author":"Li","year":"2023","journal-title":"arXiv:2305.15669"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v18i1.21959"},{"key":"ref41","first-page":"449","article-title":"A distributional perspective on reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bellemare"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"ref43","first-page":"1096","article-title":"Implicit quantile networks for distributional reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dabney"},{"key":"ref44","first-page":"4033","article-title":"Deep exploration via bootstrapped DQN","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Osband"},{"key":"ref45","first-page":"1050","article-title":"Dropout as a Bayesian approximation: Representing model uncertainty in deep learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Gal"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20855"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0749"},{"key":"ref48","article-title":"Hierarchical reinforcement learning with uncertainty-guided diffusional subgoals","author":"Huiling Wang","year":"2025","journal-title":"arXiv:2505.21750"},{"key":"ref49","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1812.05905"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20783"},{"key":"ref52","first-page":"1","article-title":"Offline RL with no OOD actionS: In-sample learning via implicit value regularization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Xu"},{"key":"ref53","first-page":"1787","article-title":"Better exploration with optimistic actor critic","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Ciosek"},{"key":"ref54","article-title":"UCB exploration via Q-ensembles","author":"Chen","year":"2017","journal-title":"arXiv:1706.01502"},{"key":"ref55","first-page":"176","article-title":"Averaged-DQN: Variance reduction and stabilization for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Anschel"},{"key":"ref56","first-page":"1","article-title":"Efficient deep reinforcement learning requires regulating overfitting","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Li"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11505906\/11267513.pdf?arnumber=11267513","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T04:40:42Z","timestamp":1777956042000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11267513\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":56,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2025.3633997","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]}}}