{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T11:41:26Z","timestamp":1770378086512,"version":"3.49.0"},"reference-count":43,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62373364,62176259"],"award-info":[{"award-number":["62373364,62176259"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/CAA J. Autom. Sinica"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1109\/jas.2025.125633","type":"journal-article","created":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T20:56:15Z","timestamp":1770152175000},"page":"57-71","source":"Crossref","is-referenced-by-count":0,"title":["Offline Generalized Actor-Critic with Distance Regularization"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2538-4761","authenticated-orcid":false,"given":"Huanting","family":"Feng","sequence":"first","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, School of Information and Control Engineering, China University of Mining and Technology,Xuzhou,China,221116"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0636-5820","authenticated-orcid":false,"given":"Yuhu","family":"Cheng","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, School of Information and Control Engineering, China University of Mining and Technology,Xuzhou,China,221116"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5327-1088","authenticated-orcid":false,"given":"Xuesong","family":"Wang","sequence":"additional","affiliation":[{"name":"Engineering Research Center of Intelligent Control for Underground Space, Ministry of Education, School of Information and Control Engineering, China University of Mining and Technology,Xuzhou,China,221116"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2023.3274908"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1111\/mafi.12382"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3098985"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2021.3054625"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2023.123705"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2021.1004272"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2023.123477"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3477600"},{"key":"ref10","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv preprint"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3250269"},{"key":"ref12","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. 36th Int. Canf. Machine Learning","author":"Fujimoto"},{"key":"ref13","article-title":"Behavior regularized offline reinforcement learning","volume-title":"Proc. Int. Conf. Learning Representations","author":"Wu"},{"key":"ref14","first-page":"7768","article-title":"Critic regularized regression","volume-title":"Proc. 34th Int. Canf. Neural Information Processing Systems","author":"Wang"},{"key":"ref15","article-title":"Advantage weighted regression: Simple and scalable off-policy reinforcement learning","volume-title":"Proc. Int. Canf. Learning Representations","author":"Peng"},{"key":"ref16","article-title":"AWAC: Accelerating online reinforcement learning with offline datasets","volume-title":"Proc. Int. Conf. Learning Representations","author":"Nair"},{"key":"ref17","first-page":"11784","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","volume-title":"Proc. 33rd Int. Conf. Neural Information Processing Systems","author":"Kumar"},{"key":"ref18","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. 35th Int. Conf. Neural Information Processing Systems, Virtual Event","author":"Fujimoto","year":"2021"},{"key":"ref19","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. 35th Int. Canf. Machine Learning","author":"Fujimoto"},{"key":"ref20","article-title":"Diffusion policies as an expressive policy class for offline reinforcement learning","volume-title":"Proc. 11 th Int. Conf. Learning Representations","author":"Wang"},{"key":"ref21","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. 34th Int. Conf. Neural Information Processing Systems","author":"Kumar"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20855"},{"key":"ref23","first-page":"18267","article-title":"Why so pessimistic? Estimating uncertainties for offline RL through ensembles, and why their independence matters","volume-title":"Proc. 36th Int. Conf. Neural Information Processing Systems","author":"Ghasemipour"},{"key":"ref24","first-page":"5774","article-title":"Offline reinforcement learning with Fisher divergence critic regularization","volume-title":"Proc. 38th Int. Conf. Machine Learning","author":"Kostrikov"},{"key":"ref25","article-title":"Offline reinforcement learning with implicit q-learning","volume-title":"Proc. 10th Int. Conf. Learning Representations","author":"Kostrikov"},{"key":"ref26","article-title":"Offline reinforcement learning with value-based episodic memory","volume-title":"Proc. 10th Int. Conf. Learning Representations","author":"Ma"},{"key":"ref27","article-title":"Extreme Q-learning: Maxent RL without entropy","volume-title":"Proc. 11th Int. Conf. Learning Representations","author":"Garg"},{"key":"ref28","article-title":"Offline RL with no OOD actions: In-sample learning via implicit value regularization","volume-title":"Proc. 11th Int. Conf. Learning Representations","author":"Xu"},{"key":"ref29","article-title":"When data geometry meets deep function: Generalizing offline reinforcement learning","volume-title":"Proc. 11th Int. Conf. Learning Representations","author":"Li"},{"key":"ref30","first-page":"28701","article-title":"Policy regularization with dataset constraint for offline reinforcement learning","volume-title":"Proc. 40th Int. Conf. Machine Learning","author":"Ran"},{"key":"ref31","first-page":"1711","article-title":"Mildly conservative Q-learning for offline reinforcement learning","volume-title":"Proc. 36th Int. Conf. Neural Information Processing Systems","author":"Lyu"},{"key":"ref32","article-title":"Pessimistic bootstrapping for uncertainty-driven offline reinforcement learning","volume-title":"Proc. 10th Int. Conf. Learning Representations","author":"Bai"},{"key":"ref33","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124227"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3051456"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2021.1003814"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.3023127"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3213566"},{"key":"ref39","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","volume-title":"Proc. Int. Conf. Learning Representations","author":"Fu"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2022.3170355"},{"key":"ref41","first-page":"11319","article-title":"Uncertainty weighted actor-critic for offline reinforcement learning","volume-title":"Proc. 38th Int. Conf. Machine Learning","author":"Wu"},{"key":"ref42","first-page":"30997","article-title":"CORL: Research-oriented deep offline reinforcement learning library","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Tarasov"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3309906"}],"container-title":["IEEE\/CAA Journal of Automatica Sinica"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6570654\/11369891\/11369926.pdf?arnumber=11369926","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,5]],"date-time":"2026-02-05T20:42:06Z","timestamp":1770324126000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11369926\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":43,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/jas.2025.125633","relation":{},"ISSN":["2329-9266","2329-9274"],"issn-type":[{"value":"2329-9266","type":"print"},{"value":"2329-9274","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1]]}}}