{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T13:15:08Z","timestamp":1783602908408,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":56,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819533428","type":"print"},{"value":"9789819533435","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,23]],"date-time":"2025-11-23T00:00:00Z","timestamp":1763856000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,23]],"date-time":"2025-11-23T00:00:00Z","timestamp":1763856000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-3343-5_46","type":"book-chapter","created":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T06:30:57Z","timestamp":1763793057000},"page":"594-606","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving RL Exploration for\u00a0LLM Reasoning Through Retrospective Replay"],"prefix":"10.1007","author":[{"given":"Shihan","family":"Dou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muling","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingwen","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rui","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Gui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,23]]},"reference":[{"issue":"6","key":"46_CR1","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/MSP.2017.2743240","volume":"34","author":"K Arulkumaran","year":"2017","unstructured":"Arulkumaran, K., Deisenroth, M.P., Brundage, M., Bharath, A.A.: Deep reinforcement learning: a brief survey. IEEE Signal Process. Mag. 34(6), 26\u201338 (2017). https:\/\/doi.org\/10.1109\/MSP.2017.2743240","journal-title":"IEEE Signal Process. Mag."},{"key":"46_CR2","doi-asserted-by":"publisher","unstructured":"Bai, Y., et al.: Training a helpful and harmless assistant with reinforcement learning from human feedback. CoRR abs\/2204.05862 (2022). https:\/\/doi.org\/10.48550\/arXiv.2204.05862","DOI":"10.48550\/arXiv.2204.05862"},{"key":"46_CR3","unstructured":"Bai, Y., et al.: Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)"},{"key":"46_CR4","unstructured":"Bi, X., et al.: Deepseek LLM: scaling open-source language models with longtermism. arXiv preprint arXiv:2401.02954 (2024)"},{"key":"46_CR5","doi-asserted-by":"crossref","unstructured":"Chai, M., et al.: Docfusion: a unified framework for document parsing tasks. arXiv preprint arXiv:2412.12505 (2024)","DOI":"10.18653\/v1\/2025.findings-acl.393"},{"key":"46_CR6","unstructured":"Christopoulou, F., et al.: Pangu-coder: program synthesis with function-level language modeling. arXiv preprint arXiv:2207.11280 (2022)"},{"key":"46_CR7","unstructured":"Cobbe, K., et al.: Training verifiers to solve math word problems. arXiv abs\/2110.14168 (2021). https:\/\/api.semanticscholar.org\/CorpusID:239998651"},{"key":"46_CR8","doi-asserted-by":"publisher","unstructured":"Dimitrakakis, C.: Tree exploration for Bayesian RL exploration. In: 2008 International Conference on Computational Intelligence for Modelling Control Automation, pp. 1029\u20131034 (2008). https:\/\/doi.org\/10.1109\/CIMCA.2008.32","DOI":"10.1109\/CIMCA.2008.32"},{"key":"46_CR9","unstructured":"Ding, Y., et al.: Mitigating tail narrowing in LLM self-improvement via socratic-guided sampling. arXiv abs\/2411.00750 (2024). https:\/\/api.semanticscholar.org\/CorpusID:273798221"},{"key":"46_CR10","unstructured":"Dou, S., et al.: What\u2019s wrong with your code generated by large language models? An extensive study. arXiv preprint arXiv:2407.06153 (2024)"},{"key":"46_CR11","doi-asserted-by":"crossref","unstructured":"Dou, S., et al.: Stepcoder: improving code generation with reinforcement learning from compiler feedback. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 4571\u20134585 (2024)","DOI":"10.18653\/v1\/2024.acl-long.251"},{"key":"46_CR12","unstructured":"Dou, S., et al.: Metarm: shifted distributions alignment via meta-learning. arXiv preprint arXiv:2405.00438 (2024)"},{"key":"46_CR13","unstructured":"Dou, S., et al.: Towards understanding the capability of large language models on code clone detection: a survey. arXiv preprint arXiv:2308.01191 (2023)"},{"key":"46_CR14","unstructured":"Dou, S., et al.: Evalearn: quantifying the learning capability and efficiency of LLMs via sequential problem solving. arXiv preprint arXiv:2506.02672 (2025)"},{"key":"46_CR15","doi-asserted-by":"crossref","unstructured":"Dou, S., et al.: Loramoe: alleviating world knowledge forgetting in large language models via moe-style plugin. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 1932\u20131945 (2024)","DOI":"10.18653\/v1\/2024.acl-long.106"},{"key":"46_CR16","doi-asserted-by":"crossref","unstructured":"Ecoffet, A., Huizinga, J., Lehman, J., Stanley, K.O., Clune, J.: First return, then explore. Nature 590, 580\u2013586 (2020). https:\/\/api.semanticscholar.org\/CorpusID:216552951","DOI":"10.1038\/s41586-020-03157-9"},{"key":"46_CR17","unstructured":"Gao, S., et al.: Linear alignment: a closed-form solution for aligning human preferences without tuning and feedback. In: Proceedings of the 41st International Conference on Machine Learning, pp. 14702\u201314722 (2024)"},{"key":"46_CR18","unstructured":"Guo, D., et al.: Deepseek-r1: incentivizing reasoning capability in LLMs via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)"},{"key":"46_CR19","unstructured":"Guo, D., et al.: Deepseek-coder: when the large language model meets programming \u2013 the rise of code intelligence (2024). https:\/\/api.semanticscholar.org\/CorpusID:267211867"},{"key":"46_CR20","unstructured":"Hendrycks, D., et al.: Measuring coding challenge competence with APPS. In: Vanschoren, J., Yeung, S. (eds.) Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks 1, NeurIPS Datasets and Benchmarks 2021, December 2021, virtual (2021). https:\/\/datasets-benchmarks-proceedings.neurips.cc\/paper\/2021\/hash\/c24cd76e1ce41366a4bbe8a49b02a028-Abstract-round2.html"},{"key":"46_CR21","unstructured":"Hendrycks, D., et al.: Measuring mathematical problem solving with the math dataset. arXiv abs\/2103.03874 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232134851"},{"key":"46_CR22","doi-asserted-by":"crossref","unstructured":"Huang, J., Chang, K.C.C.: Towards reasoning in large language models: a survey. arXiv preprint arXiv:2212.10403 (2022)","DOI":"10.18653\/v1\/2023.findings-acl.67"},{"key":"46_CR23","unstructured":"Jiang, D., et al.: Rationalyst: pre-training process-supervision for improving reasoning. arXiv preprint arXiv:2410.01044 (2024)"},{"key":"46_CR24","unstructured":"Kirk, R., et al.: Understanding the effects of RLHF on LLM generalisation and diversity. arXiv preprint arXiv:2310.06452 (2023)"},{"key":"46_CR25","unstructured":"Kumar, K., et al.: LLM post-training: a deep dive into reasoning large language models. arXiv preprint arXiv:2502.21321 (2025)"},{"key":"46_CR26","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.inffus.2022.03.003","volume":"85","author":"P Ladosz","year":"2022","unstructured":"Ladosz, P., Weng, L., Kim, M., Oh, H.: Exploration in deep reinforcement learning: a survey. Inf. Fusion 85, 1\u201322 (2022)","journal-title":"Inf. Fusion"},{"key":"46_CR27","unstructured":"Li, R., et al.: Starcoder: may the source be with you! arXiv preprint arXiv:2305.06161 (2023)"},{"key":"46_CR28","unstructured":"Li, Y.: Deep reinforcement learning: an overview. arXiv preprint arXiv:1701.07274 (2017)"},{"key":"46_CR29","unstructured":"Liu, J., et al.: RLTF: reinforcement learning from unit test feedback. arXiv preprint arXiv:2307.04349 (2023)"},{"key":"46_CR30","unstructured":"Luo, Z., et al.: Wizardcoder: empowering code large language models with evol-instruct. arXiv preprint arXiv:2306.08568 (2023)"},{"key":"46_CR31","doi-asserted-by":"crossref","unstructured":"Mercer, S., Spillard, S., Martin, D.P.: Brief analysis of deepseek R1 and it\u2019s implications for generative AI. arXiv preprint arXiv:2502.02523 (2025)","DOI":"10.70777\/si.v2i1.11097"},{"key":"46_CR32","doi-asserted-by":"publisher","unstructured":"Nair, A., McGrew, B., Andrychowicz, M., Zaremba, W., Abbeel, P.: Overcoming exploration in reinforcement learning with demonstrations. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 6292\u20136299 (2018). https:\/\/doi.org\/10.1109\/ICRA.2018.8463162","DOI":"10.1109\/ICRA.2018.8463162"},{"key":"46_CR33","doi-asserted-by":"crossref","unstructured":"Plaat, A., Wong, A., Verberne, S., Broekens, J., van Stein, N., Back, T.: Reasoning with large language models, a survey. arXiv preprint arXiv:2407.11511 (2024)","DOI":"10.1145\/3774896"},{"key":"46_CR34","unstructured":"Roziere, B., et al.: Code llama: open foundation models for code. arXiv preprint arXiv:2308.12950 (2023)"},{"key":"46_CR35","unstructured":"Schaul, T., Quan, J., Antonoglou, I., Silver, D.: Prioritized experience replay. arXiv preprint arXiv:1511.05952 (2015)"},{"key":"46_CR36","unstructured":"Schulman, J., Moritz, P., Levine, S., Jordan, M., Abbeel, P.: High-dimensional continuous control using generalized advantage estimation. arXiv preprint arXiv:1506.02438 (2015)"},{"key":"46_CR37","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"46_CR38","unstructured":"Seed, B., et al.: Seed1. 5-thinking: advancing superb reasoning models with reinforcement learning. arXiv preprint arXiv:2504.13914 (2025)"},{"key":"46_CR39","unstructured":"Shao, Z., et al.: Deepseekmath: pushing the limits of mathematical reasoning in open language models. arXiv abs\/2402.03300 (2024). https:\/\/api.semanticscholar.org\/CorpusID:267412607"},{"key":"46_CR40","doi-asserted-by":"crossref","unstructured":"Shen, W., et al.: Loose lips sink ships: mitigating length bias in reinforcement learning from human feedback. In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.188"},{"key":"46_CR41","unstructured":"Shojaee, P., Jain, A., Tipirneni, S., Reddy, C.K.: Execution-based code generation using deep reinforcement learning. arXiv preprint arXiv:2301.13816 (2023)"},{"key":"46_CR42","unstructured":"Sutton, R.S., McAllester, D., Singh, S., Mansour, Y.: Policy gradient methods for reinforcement learning with function approximation. In: Advances in Neural Information Processing Systems, vol. 12 (1999)"},{"key":"46_CR43","unstructured":"Tao, S., Shukla, A., Chan, T.k., Su, H.: Reverse forward curriculum learning for extreme sample and demonstration efficiency in reinforcement learning. arXiv preprint arXiv:2405.03379 (2024)"},{"key":"46_CR44","unstructured":"Thrun, S.B.: Efficient exploration in reinforcement learning. Technical report, USA (1992)"},{"key":"46_CR45","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"46_CR46","unstructured":"Uesato, J., et al.: Solving math word problems with process-and outcome-based feedback. arXiv preprint arXiv:2211.14275 (2022)"},{"key":"46_CR47","unstructured":"Wang, B., et al.: Secrets of RLHF in large language models part ii: reward modeling. arXiv preprint arXiv:2401.06080 (2024)"},{"key":"46_CR48","doi-asserted-by":"crossref","unstructured":"Wen, L., et al.: Light-r1: curriculum SFT, DPO and RL for long cot from scratch and beyond. arXiv preprint arXiv:2503.10460 (2025)","DOI":"10.18653\/v1\/2025.acl-industry.24"},{"key":"46_CR49","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1023\/A:1022672621406","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach. Learn. 8, 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"key":"46_CR50","unstructured":"Wu, M., et al.: Progressive mastery: customized curriculum learning with guided prompting for mathematical reasoning. arXiv preprint arXiv:2506.04065 (2025)"},{"key":"46_CR51","unstructured":"Xi, Z., et al.: Training large language models for reasoning through reverse curriculum reinforcement learning. In: International Conference on Machine Learning, pp. 54030\u201354048. PMLR (2024)"},{"key":"46_CR52","unstructured":"Yang, T., et al.: Exploration in deep reinforcement learning: a comprehensive survey. arXiv preprint arXiv:2109.06668 (2021)"},{"key":"46_CR53","unstructured":"Zang, J., et al.: Compression hacking: a supplementary perspective on informatics metric of language models from geometric distortion. arXiv preprint arXiv:2505.17793 (2025)"},{"key":"46_CR54","unstructured":"Zhang, S., Sutton, R.S.: A deeper look at experience replay. arXiv preprint arXiv:1712.01275 (2017)"},{"key":"46_CR55","unstructured":"Zhao, W.X., et al.: A survey of large language models"},{"key":"46_CR56","unstructured":"Zheng, R., et al.: Delve into PPO: implementation matters for stable RLHF. In: NeurIPS 2023 Workshop on Instruction Tuning and Instruction Following (2023)"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-3343-5_46","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T12:30:44Z","timestamp":1783600244000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-3343-5_46"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,23]]},"ISBN":["9789819533428","9789819533435"],"references-count":56,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-3343-5_46","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,23]]},"assertion":[{"value":"23 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}