{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T08:08:01Z","timestamp":1764144481309,"version":"3.46.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2018AAA0100802"],"award-info":[{"award-number":["2018AAA0100802"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s10489-025-06972-7","type":"journal-article","created":{"date-parts":[[2025,11,2]],"date-time":"2025-11-02T21:25:15Z","timestamp":1762118715000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Offline-to-online reinforcement learning with policy ensemble and policy-extended value"],"prefix":"10.1007","volume":"55","author":[{"given":"Jiacheng","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6038-4339","authenticated-orcid":false,"given":"Jin","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,2]]},"reference":[{"issue":"7","key":"6972_CR1","doi-asserted-by":"publisher","first-page":"10961","DOI":"10.1007\/s11042-022-13695-1","volume":"82","author":"J Montalvo","year":"2023","unstructured":"Montalvo J, Garc\u00eda-Mart\u00edn \u00c1, Besc\u00f3s J (2023) Exploiting semantic segmentation to boost reinforcement learning in video game environments. Multimed Tools Appl 82(7):10961\u201310979","journal-title":"Multimed Tools Appl"},{"issue":"1","key":"6972_CR2","first-page":"355","volume":"15","author":"G Rani","year":"2023","unstructured":"Rani G, Pandey U, Wagde AA, Dhaka VS (2023) A deep reinforcement learning technique for bug detection in video games. Int J Inf Technol 15(1):355\u2013367","journal-title":"Int J Inf Technol"},{"issue":"24","key":"6972_CR3","doi-asserted-by":"publisher","first-page":"30677","DOI":"10.1007\/s10489-023-05156-5","volume":"53","author":"C Chen","year":"2023","unstructured":"Chen C, Zhang C, Pan Y (2023) Active compliance control of robot peg-in-hole assembly based on combined reinforcement learning. Appl Intell 53(24):30677\u201330690","journal-title":"Appl Intell"},{"issue":"1","key":"6972_CR4","doi-asserted-by":"publisher","first-page":"244","DOI":"10.1109\/TAI.2023.3237665","volume":"5","author":"X He","year":"2023","unstructured":"He X, Lv C (2023) Robotic control in adversarial and sparse reward environments: a robust goal-conditioned reinforcement learning approach. IEEE Trans Artif Intell 5(1):244\u2013253","journal-title":"IEEE Trans Artif Intell"},{"key":"6972_CR5","doi-asserted-by":"crossref","unstructured":"Yu Y (2018) Towards sample efficient reinforcement learning. In: IJCAI, pp 5739\u20135743","DOI":"10.24963\/ijcai.2018\/820"},{"issue":"22","key":"6972_CR6","doi-asserted-by":"publisher","first-page":"27128","DOI":"10.1007\/s10489-023-04911-y","volume":"53","author":"Z Sun","year":"2023","unstructured":"Sun Z, Jing C, Guo S, An L (2023) Pac-bayesian offline meta-reinforcement learning. Appl Intell 53(22):27128\u201327147","journal-title":"Appl Intell"},{"issue":"23","key":"6972_CR7","doi-asserted-by":"publisher","first-page":"12156","DOI":"10.1007\/s10489-024-05830-2","volume":"54","author":"B Xia","year":"2024","unstructured":"Xia B, Yang Z, Xie M, Chang Y, Yuan B, Li Z, Wang X, Liang B (2024) Solving time-delay issues in reinforcement learning via transformers. Appl Intell 54(23):12156\u201312176","journal-title":"Appl Intell"},{"key":"6972_CR8","unstructured":"Kumar A, Fu J, Soh M, Tucker G, Levine S (2019) Stabilizing off-policy q-learning via bootstrapping error reduction. Adv Neural Inf Process Syst 32"},{"key":"6972_CR9","unstructured":"Prudencio RF, Maximo MR, Colombini EL (2023) A survey on offline reinforcement learning: taxonomy, review, and open problems. IEEE Trans Neural Netw Learn Syst"},{"key":"6972_CR10","unstructured":"Nair A, Gupta A, Dalal M, Levine S (2020) Awac: Accelerating online reinforcement learning with offline datasets. arXiv:2006.09359"},{"key":"6972_CR11","unstructured":"Lee S, Seo Y, Lee K, Abbeel P, Shin J (2022) Offline-to-online reinforcement learning via balanced replay and pessimistic q-ensemble. In: Conference on robot learning. PMLR, pp 1702\u20131712"},{"key":"6972_CR12","doi-asserted-by":"crossref","unstructured":"Zhao Y, Boney R, Ilin A, Kannala J, Pajarinen J (2022) Adaptive behavior cloning regularization for stable offline-to-online reinforcement learning. In: Proceedings of the European Symposium on Artificial Neural Networks. European Symposium on Artificial Neural Networks (ESANN)","DOI":"10.14428\/esann\/2022.ES2022-110"},{"key":"6972_CR13","unstructured":"Uchendu I, Xiao T, Lu Y, Zhu B, Yan M, Simon J, Bennice M, Fu C, Ma C, Jiao J et al (2023) Jump-start reinforcement learning. In: International conference on machine learning. PMLR, pp 34556\u201334583"},{"key":"6972_CR14","unstructured":"Kostrikov I, Nair A, Levine S (2022) Offline reinforcement learning with implicit q-learning. In: International conference on learning representations"},{"key":"6972_CR15","unstructured":"Zheng Q, Zhang A, Grover A (2022) Online decision transformer. In: International conference on machine learning. PMLR, pp 27042\u201327059"},{"key":"6972_CR16","first-page":"31278","volume":"35","author":"J Wu","year":"2022","unstructured":"Wu J, Wu H, Qiu Z, Wang J, Long M (2022) Supported policy optimization for offline reinforcement learning. Adv Neural Inf Process Syst 35:31278\u201331291","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR17","unstructured":"Mark MS, Ghadirzadeh A, Chen X, Finn C (2022) Fine-tuning offline policies with optimistic action selection. In: Deep reinforcement learning workshop NeurIPS 2022"},{"key":"6972_CR18","unstructured":"Mao Y, Wang C, Wang B, Zhang C (2022) Moore: Model-based offline-to-online reinforcement learning. arXiv:2201.10070"},{"key":"6972_CR19","unstructured":"Zhao K, Ma Y, Liu J, Zheng Y, Meng Z (2023) Ensemble-based offline-to-online reinforcement learning: from pessimistic learning to optimistic exploration. arXiv:2306.06871"},{"key":"6972_CR20","unstructured":"Nakamoto M, Zhai S, Singh A, Sobol Mark M, Ma Y, Finn C, Kumar A, Levine S (2024) Cal-ql: Calibrated offline rl pre-training for efficient online fine-tuning. Adv Neural Inf Process Syst 36"},{"key":"6972_CR21","unstructured":"Li J, Hu X, Xu H, Liu J, Zhan X, Zhang Y-Q (2023) Proto: Iterative policy regularized offline-to-online reinforcement learning. arXiv:2305.15669"},{"key":"6972_CR22","doi-asserted-by":"crossref","unstructured":"Guo S, Zou L, Chen H, Qu B, Chi H, Philip SY, Chang Y (2023) Sample efficient offline-to-online reinforcement learning. IEEE Trans Knowl Data Eng","DOI":"10.1109\/TKDE.2023.3302804"},{"key":"6972_CR23","unstructured":"Zhang H, Xu W, Yu H (2023) Policy expansion for bridging offline-to-online reinforcement learning. In: The Eleventh International Conference on Learning Representations"},{"key":"6972_CR24","first-page":"27537","volume":"35","author":"J Zhang","year":"2022","unstructured":"Zhang J, Li S, Zhang C (2022) Cup: Critic-guided policy reuse. Adv Neural Inf Process Syst 35:27537\u201327548","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR25","unstructured":"Lei K, He Z, Lu C, Hu K, Gao Y, Xu H (2024) Uni-o4: Unifying online and offline deep reinforcement learning with multi-step on-policy optimization. In: The Twelfth International Conference on Learning Representations"},{"key":"6972_CR26","unstructured":"Fujimoto S, Meger D, Precup D (2019) Off-policy deep reinforcement learning without exploration. In: International conference on machine learning. PMLR, pp 2052\u20132062"},{"key":"6972_CR27","first-page":"20132","volume":"34","author":"S Fujimoto","year":"2021","unstructured":"Fujimoto S, Gu SS (2021) A minimalist approach to offline reinforcement learning. Adv Neural Inf Process Syst 34:20132\u201320145","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR28","first-page":"1179","volume":"33","author":"A Kumar","year":"2020","unstructured":"Kumar A, Zhou A, Tucker G, Levine S (2020) Conservative q-learning for offline reinforcement learning. Adv Neural Inf Process Syst 33:1179\u20131191","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR29","first-page":"7436","volume":"34","author":"G An","year":"2021","unstructured":"An G, Moon S, Kim J-H, Song HO (2021) Uncertainty-based offline reinforcement learning with diversified q-ensemble. Adv Neural Inf Process Syst 34:7436\u20137447","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR30","unstructured":"Bai C, Wang L, Yang Z, Deng Z-H, Garg A, Liu P, Wang Z (2022) Pessimistic bootstrapping for uncertainty-driven offline reinforcement learning. In: International Conference on Learning Representations"},{"key":"6972_CR31","first-page":"36902","volume":"35","author":"X Chen","year":"2022","unstructured":"Chen X, Ghadirzadeh A, Yu T, Wang J, Gao AY, Li W, Bin L, Finn C, Zhang C (2022) Lapo: Latent-variable advantage-weighted policy optimization for offline reinforcement learning. Adv Neural Inf Process Syst 35:36902\u201336913","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR32","first-page":"14129","volume":"33","author":"T Yu","year":"2020","unstructured":"Yu T, Thomas G, Yu L, Ermon S, Zou JY, Levine S, Finn C, Ma T (2020) Mopo: Model-based offline policy optimization. Adv Neural Inf Process Syst 33:14129\u201314142","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR33","first-page":"21810","volume":"33","author":"R Kidambi","year":"2020","unstructured":"Kidambi R, Rajeswaran A, Netrapalli P, Joachims T (2020) Morel: Model-based offline reinforcement learning. Adv Neural Inf Process Syst 33:21810\u201321823","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR34","unstructured":"Lu Y, Hausman K, Chebotar Y, Yan M, Jang E, Herzog A, Xiao T, Irpan A, Khansari M, Kalashnikov D et al (2022) Aw-opt: Learning robotic skills with imitation andreinforcement at scale. In: Conference on robot learning. PMLR, pp 1078\u20131088"},{"key":"6972_CR35","first-page":"36599","volume":"35","author":"H Niu","year":"2022","unstructured":"Niu H, Qiu Y, Li M, Zhou G, Hu J, Zhan X et al (2022) When to trust your simulator: dynamics-aware hybrid offline-and-online reinforcement learning. Adv Neural Inf Process Syst 35:36599\u201336612","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR36","unstructured":"Zhao K, Ma Y, Liu J, Hao J, Zheng Y, Meng Z (2023) Improving offline-to-online reinforcement learning with q-ensembles. In: ICML Workshop on new frontiers in learning, control, and dynamical systems"},{"key":"6972_CR37","doi-asserted-by":"crossref","unstructured":"Zhang Y, Liu J, Li C, Niu Y, Yang Y, Liu Y, Ouyang W (2024) A perspective of q-value estimation on offline-to-online reinforcement learning. In: AAAI, pp 16908\u201316916","DOI":"10.1609\/aaai.v38i15.29633"},{"key":"6972_CR38","unstructured":"Zhou Z, Peng A, Li Q, Levine S, Kumar A (2025) Efficient online reinforcement learning fine-tuning need not retain offline data. In: The Thirteenth International Conference on Learning Representations"},{"key":"6972_CR39","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1016\/j.neucom.2020.02.117","volume":"392","author":"Y Wang","year":"2020","unstructured":"Wang Y, Liu Y, Chen W, Ma Z-M, Liu T-Y (2020) Target transfer q-learning and its convergence analysis. Neurocomputing 392:11\u201322","journal-title":"Neurocomputing"},{"key":"6972_CR40","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv:1707.06347"},{"key":"6972_CR41","unstructured":"Zhuang Z, Lei K, Liu J, Wang D, Guo Y (2023) Behavior proximal policy optimization. In: The Eleventh International Conference on Learning Representations"},{"key":"6972_CR42","first-page":"25502","volume":"34","author":"D Ghosh","year":"2021","unstructured":"Ghosh D, Rahme J, Kumar A, Zhang A, Adams RP, Levine S (2021) Why generalization in rl is difficult: epistemic pomdps and implicit partial observability. Adv Neural Inf Process Syst 34:25502\u201325515","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR43","doi-asserted-by":"crossref","unstructured":"Yang Z, Ren K, Luo X, Liu M, Liu W, Bian J, Zhang W, Li D (2022) Towards applicable reinforcement learning: improving the generalization and sample efficiency with policy ensemble. arXiv:2205.09284","DOI":"10.24963\/ijcai.2022\/508"},{"key":"6972_CR44","unstructured":"Lee K, Laskin M, Srinivas A, Abbeel P (2021) Sunrise: a simple unified framework for ensemble learning in deep reinforcement learning. In: International conference on machine learning. PMLR, pp 6131\u20136141"},{"key":"6972_CR45","doi-asserted-by":"publisher","first-page":"126381","DOI":"10.1016\/j.neucom.2023.126381","volume":"548","author":"G Liu","year":"2023","unstructured":"Liu G, Chen G, Huang V (2023) Policy ensemble gradient for continuous control problems in deep reinforcement learning. Neurocomputing 548:126381","journal-title":"Neurocomputing"},{"key":"6972_CR46","doi-asserted-by":"crossref","unstructured":"Tang H, Meng Z, Hao J, Chen C, Graves D, Li D, Yu C, Mao H, Liu W, Yang Y et al (2022) What about inputting policy in value function: policy representation and policy-extended value function approximator. In: Proceedings of the AAAI conference on artificial intelligence, vol 36, pp 8441\u20138449","DOI":"10.1609\/aaai.v36i8.20820"},{"key":"6972_CR47","doi-asserted-by":"crossref","unstructured":"Zhao Y, Ding Y, Pei Y (2024) Adaptive optimization in evolutionary reinforcement learning using evolutionary mutation rates. IEEE Access","DOI":"10.1109\/ACCESS.2024.3493198"},{"key":"6972_CR48","unstructured":"Fu J, Kumar A, Nachum O, Tucker G, Levine S (2020) D4rl: Datasets for deep data-driven reinforcement learning. arXiv:2004.07219"},{"key":"6972_CR49","first-page":"30997","volume":"36","author":"D Tarasov","year":"2023","unstructured":"Tarasov D, Nikulin A, Akimov D, Kurenkov V, Kolesnikov S (2023) Corl: Research-oriented deep offline reinforcement learning library. Adv Neural Inf Process Syst 36:30997\u201331020","journal-title":"Adv Neural Inf Process Syst"},{"key":"6972_CR50","unstructured":"Andersen G, Vrancx P, Bou-Ammar H (2018) Learning high-level representations from demonstrations. arXiv:1802.06604"},{"key":"6972_CR51","doi-asserted-by":"crossref","unstructured":"Mazzaglia P, Catal O, Verbelen T, Dhoedt B (2022) Curiosity-driven exploration via latent bayesian surprise. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 36, pp 7752\u20137760","DOI":"10.1609\/aaai.v36i7.20743"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06972-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06972-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06972-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T08:04:08Z","timestamp":1764144248000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06972-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":51,"journal-issue":{"issue":"16","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["6972"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06972-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2025,11]]},"assertion":[{"value":"27 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 November 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"All the experiments in this paper are computer simulations of games and do not involve experiments on animals, plants, or human entities.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"The paper does not include data or images that require permission to be published.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"1067"}}