{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T23:21:11Z","timestamp":1777936871758,"version":"3.51.4"},"publisher-location":"Cham","reference-count":27,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031965586","type":"print"},{"value":"9783031965593","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-96559-3_10","type":"book-chapter","created":{"date-parts":[[2025,6,29]],"date-time":"2025-06-29T07:14:06Z","timestamp":1751181246000},"page":"142-156","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Offline-to-Online: Case-Based Knowledge Distillation with\u00a0Large Language Models for\u00a0Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Hongzhe","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meilong","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiming","family":"Cui","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,23]]},"reference":[{"key":"10_CR1","doi-asserted-by":"crossref","unstructured":"Bianchi, R.A., Ros, R., Lopez\u00a0de Mantaras, R.: Improving reinforcement learning by using case based heuristics. In: International Conference on Case-Based Reasoning, pp. 75\u201389. Springer (2009)","DOI":"10.1007\/978-3-642-02998-1_7"},{"key":"10_CR2","unstructured":"Atzeni, M., Dhuliawala, S., Murugesan, K., Sachan, M.: Case-based reasoning for better generalization in textual reinforcement learning, arXiv preprint arXiv:2110.08470 (2021)"},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Summary of chatgpt-related research and perspective towards the future of large language models. Meta-Radiol. 100017 (2023)","DOI":"10.1016\/j.metrad.2023.100017"},{"key":"10_CR4","unstructured":"Zhang, F., et al.: Improving sample efficiency of reinforcement learning with background knowledge from large language models, arXiv preprint arXiv:2407.03964 (2024)"},{"key":"10_CR5","unstructured":"Yan, X., et al.: Efficient reinforcement learning with large language model priors, arXiv preprint arXiv:2410.07927 (2024)"},{"key":"10_CR6","unstructured":"Nair, A., et al.: AWAC: accelerating online reinforcement learning with offline datasets, arXiv preprint arXiv:2006.09359 (2020)"},{"key":"10_CR7","unstructured":"Zhang, H., et al.: Policy expansion for bridging offline-to-online reinforcement learning, arXiv preprint arXiv:2302.00935 (2023)"},{"key":"10_CR8","unstructured":"Hinton, G.: Distilling the knowledge in a neural network, arXiv preprint arXiv:1503.02531 (2015)"},{"key":"10_CR9","first-page":"226","volume":"35","author":"W-C Tseng","year":"2022","unstructured":"Tseng, W.-C., et al.: Offline multi-agent reinforcement learning with knowledge distillation. Adv. Neural. Inf. Process. Syst. 35, 226\u2013237 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR10","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"511","DOI":"10.1007\/978-3-319-11209-1_36","volume-title":"Case-Based Reasoning Research and Development","author":"S Wender","year":"2014","unstructured":"Wender, S., Watson, I.: Combining case-based reasoning and reinforcement learning for unit navigation in real-time strategy game AI. In: Lamontagne, L., Plaza, E. (eds.) ICCBR 2014. LNCS (LNAI), vol. 8765, pp. 511\u2013525. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-11209-1_36"},{"key":"10_CR11","unstructured":"Lee, Y., Hu, E.S., Yang, Z., Lim, J.J.: To follow or not to follow: selective imitation learning from observations, arXiv preprint arXiv:1912.07670 (2019)"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"N\u00fc\u00dflein, J., Illium, S., M\u00fcller, R., Gabor, T., Linnhoff-Popien, C.: Case-based inverse reinforcement learning using temporal coherence. In: International Conference on Case-Based Reasoning, pp. 304\u2013317. Springer (2022)","DOI":"10.1007\/978-3-031-14923-8_20"},{"key":"10_CR13","unstructured":"Liu, Z., et al.: Knowing what not to do: Leverage language model insights for action space pruning in multi-agent reinforcement learning, arXiv preprint arXiv:2405.16854 (2024)"},{"key":"10_CR14","unstructured":"Zhou, Z., et al.: Large language model as a policy teacher for training reinforcement learning agents, arXiv preprint arXiv:2311.13373 (2023)"},{"key":"10_CR15","unstructured":"Tan, W., et al.: True knowledge comes from practice: aligning LLMs with embodied environments via reinforcement learning, arXiv preprint arXiv:2401.14151 (2024)"},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"Lee, A.X., et al.: How to spend your robot time: Bridging kickstarting and offline reinforcement learning for vision-based robotic manipulation. In: 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 2468\u20132475. IEEE (2022)","DOI":"10.1109\/IROS47612.2022.9981126"},{"key":"10_CR17","unstructured":"Xie, T., et al.: Policy finetuning: bridging sample-efficient offline and online reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a034, pp. 27395\u201327407 (2021)"},{"key":"10_CR18","unstructured":"Lei, K., et al.: Uni-o4: unifying online and offline deep reinforcement learning with multi-step on-policy optimization, arXiv preprint arXiv:2311.03351 (2023)"},{"key":"10_CR19","unstructured":"Wang, S., et al.: Train once, get a family: State-adaptive balances for offline-to-online reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a036 (2024)"},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Torrey, L., Taylor, M.: Teaching on a budget: agents advising agents in reinforcement learning. In: Proceedings of the 2013 International Conference on Autonomous Agents and Multi-Agent Systems, pp. 1053\u20131060 (2013)","DOI":"10.65109\/KZVN5966"},{"key":"10_CR21","unstructured":"Uchendu, I., et al.: Jump-start reinforcement learning. In: International Conference on Machine Learning, pp. 34556\u201334583. PMLR (2023)"},{"key":"10_CR22","unstructured":"Schmitt et al.: Kickstarting deep reinforcement learning, arXiv preprint arXiv:1803.03835 (2018)"},{"key":"10_CR23","unstructured":"Matthews et al.: Hierarchical kickstarting for skill transfer in reinforcement learning, arXiv preprint arXiv:2207.11584 (2022)"},{"key":"10_CR24","unstructured":"Schulman, J., et al.: Proximal policy optimization algorithms, arXiv preprint arXiv:1707.06347 (2017)"},{"key":"10_CR25","unstructured":"Chevalier-Boisvert et al.: Minigrid & miniworld: modular & customizable reinforcement learning environments for goal-oriented tasks. In: Advances in Neural Information Processing Systems, vol.\u00a036 (2024)"},{"key":"10_CR26","unstructured":"Du, Z., et al.: GLM: general language model pretraining with autoregressive blank infilling, arXiv preprint arXiv:2103.10360 (2021)"},{"key":"10_CR27","doi-asserted-by":"crossref","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. In: Advances in Neural Information Processing Systems, vol.\u00a035, pp. 24824\u201324837 (2022)","DOI":"10.52202\/068431-1800"}],"container-title":["Lecture Notes in Computer Science","Case-Based Reasoning Research and Development"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-96559-3_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T03:45:38Z","timestamp":1777693538000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-96559-3_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031965586","9783031965593"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-96559-3_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"23 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICCBR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Case-Based Reasoning","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Biarritz","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"33","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iccbr2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}