{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:46:30Z","timestamp":1783611990309,"version":"3.55.0"},"reference-count":38,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["623B1024"],"award-info":[{"award-number":["623B1024"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202238"],"award-info":[{"award-number":["62202238"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2024M751506"],"award-info":[{"award-number":["2024M751506"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"Collaborative Innovation Center of Novel Software Technology and Industrialization","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1109\/tnnls.2025.3589418","type":"journal-article","created":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T17:56:16Z","timestamp":1753379776000},"page":"19762-19774","source":"Crossref","is-referenced-by-count":2,"title":["Online Adaptable Offline RL With Guidance Model"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-8140-2621","authenticated-orcid":false,"given":"Xun","family":"Wang","sequence":"first","affiliation":[{"name":"State Key Laboratory for Novel Software Technology and School of Computer Science, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8296-8964","authenticated-orcid":false,"given":"Jingmian","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology and School of Computer Science, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1625-7575","authenticated-orcid":false,"given":"Zhuzhong","family":"Qian","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology and School of Computer Science, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9406-4499","authenticated-orcid":false,"given":"Bolei","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3185994"},{"key":"ref3","article-title":"Playing Atari with deep reinforcement learning","author":"Mnih","year":"2013","journal-title":"arXiv:1312.5602"},{"key":"ref4","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NIPS)","author":"Ouyang"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3054625"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3477600"},{"key":"ref7","article-title":"MT-opt: Continuous multi-task robotic reinforcement learning at scale","author":"Kalashnikov","year":"2021","journal-title":"arXiv:2104.08212"},{"key":"ref8","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv:2005.01643"},{"key":"ref9","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Fujimoto"},{"key":"ref10","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kumar"},{"key":"ref11","first-page":"28954","article-title":"Combo: Conservative offline model-based policy optimization","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Yu"},{"key":"ref12","article-title":"RAMBO-RL: Robust adversarial model-based offline reinforcement learning","author":"Rigter","year":"2022","journal-title":"arXiv:2204.12581"},{"key":"ref13","first-page":"1702","article-title":"Offline-to-online reinforcement learning via balanced replay and pessimistic Q-ensemble","volume-title":"Proc. Conf. Robot Learn.","author":"Lee"},{"key":"ref14","first-page":"62244","article-title":"Cal-QL: Calibrated offline RL pre-training for efficient online fine-tuning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Nakamoto"},{"key":"ref15","first-page":"23190","article-title":"Understanding plasticity in neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lyle"},{"key":"ref16","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020","journal-title":"arXiv:2004.07219"},{"key":"ref17","first-page":"14129","article-title":"MOPO: Model-based offline policy optimization","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Yu"},{"key":"ref18","first-page":"21810","article-title":"MOReL: Model-based offline reinforcement learning","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kidambi"},{"key":"ref19","first-page":"12519","article-title":"When to trust your model: Model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"J\u00e4nner"},{"key":"ref20","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref21","article-title":"Offline reinforcement learning with implicit Q-Learning","author":"Kostrikov","year":"2021","journal-title":"arXiv:2110.06169"},{"key":"ref22","first-page":"4085","article-title":"A policy-guided imitation approach for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.3233\/FAIA230597"},{"key":"ref24","article-title":"On integral probability metrics, \u03a6-divergences and binary classification","author":"Sriperumbudur","year":"2009","journal-title":"arXiv:0901.2698"},{"key":"ref25","first-page":"1089","article-title":"Estimating divergence functionals and the likelihood ratio by penalized convex risk minimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"20","author":"Nguyen"},{"key":"ref26","article-title":"Neorl-2: Near real-world benchmarks for offline reinforcement learning with extended realistic scenarios","author":"Gao","year":"2025","journal-title":"arXiv:2503.19267"},{"issue":"19","key":"ref27","first-page":"20033","article-title":"SUMO: Search-based uncertainty estimation for model-based offline reinforcement learning","volume-title":"Proc. AAAI Conf. Artif. Intell","volume":"39","author":"Qiao"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3497667"},{"key":"ref29","article-title":"AWAC: Accelerating online reinforcement learning with offline datasets","author":"Nair","year":"2020","journal-title":"arXiv:2006.09359"},{"key":"ref30","article-title":"Policy expansion for bridging offline-to-online reinforcement learning","author":"Zhang","year":"2023","journal-title":"arXiv:2302.00935"},{"issue":"86","key":"ref31","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."},{"key":"ref32","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","volume":"97","author":"Fujimoto"},{"key":"ref33","first-page":"11784","article-title":"Stabilizing off-policy Q-Learning via bootstrapping error reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kumar"},{"key":"ref34","article-title":"Behavior regularized offline reinforcement learning","author":"Wu","year":"2019","journal-title":"arXiv:1911.11361"},{"key":"ref35","first-page":"5774","article-title":"Offline reinforcement learning with Fisher divergence critic regularization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kostrikov"},{"key":"ref36","first-page":"1711","article-title":"Mildly conservative Q-Learning for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lyu"},{"key":"ref37","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified Q-Ensemble","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"An"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3309906"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11220834\/11095835.pdf?arnumber=11095835","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T18:03:27Z","timestamp":1761847407000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11095835\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":38,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2025.3589418","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11]]}}}