{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:12:04Z","timestamp":1783786324636,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819228553","type":"print"},{"value":"9789819228560","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2856-0_45","type":"book-chapter","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:18:46Z","timestamp":1783783126000},"page":"629-640","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Exp$$^{2}$$RL: Enhancing LLM Agent Training with\u00a0Expert Experiences"],"prefix":"10.1007","author":[{"given":"Yicheng","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qianglong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhirui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Gong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,12]]},"reference":[{"key":"45_CR1","doi-asserted-by":"publisher","unstructured":"Bayer, M.: ActiveLLM: large language model-based active learning for textual few-shot scenarios. In: Bayer, M. (ed.) Deep Learning in Textual Low-Data Regimes for Cybersecurity, pp. 89\u2013112. Springer Fachmedien, Wiesbaden (2025). https:\/\/doi.org\/10.1007\/978-3-658-48778-2_7","DOI":"10.1007\/978-3-658-48778-2_7"},{"key":"45_CR2","doi-asserted-by":"publisher","unstructured":"Chen, K., et al.: Reinforcement Learning for Long-Horizon Interactive LLM Agents. CoRR arxiv:2502.01600 (2025). https:\/\/doi.org\/10.48550\/ARXIV.2502.01600","DOI":"10.48550\/ARXIV.2502.01600"},{"key":"45_CR3","doi-asserted-by":"publisher","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models (2021). https:\/\/doi.org\/10.48550\/arXiv.2106.09685","DOI":"10.48550\/arXiv.2106.09685"},{"key":"45_CR4","doi-asserted-by":"publisher","unstructured":"Jimenez, C.E., et al.: SWE-bench: can language models resolve real-world github issues? (2024). https:\/\/doi.org\/10.48550\/arXiv.2310.06770","DOI":"10.48550\/arXiv.2310.06770"},{"key":"45_CR5","doi-asserted-by":"publisher","unstructured":"Li, D., Zhang, Y., Wang, Z., Tan, S., Kosugi, S., Okumura, M.: Active learning for abstractive text summarization via LLM-determined curriculum and certainty gain maximization. In: Al-Onaizan, Y., Bansal, M., Chen, Y.N. (eds.) Findings of the Association for Computational Linguistics: EMNLP 2024, pp. 8959\u20138971. Association for Computational Linguistics, Miami (2024). https:\/\/doi.org\/10.18653\/v1\/2024.findings-emnlp.523","DOI":"10.18653\/v1\/2024.findings-emnlp.523"},{"key":"45_CR6","doi-asserted-by":"publisher","unstructured":"Li, Z., et al.: Knapsack RL: unlocking exploration of LLMs via optimizing budget allocation (2025). https:\/\/doi.org\/10.48550\/arXiv.2509.25849","DOI":"10.48550\/arXiv.2509.25849"},{"key":"45_CR7","doi-asserted-by":"publisher","unstructured":"Liu, P., Shi, C., Sun, W.W.: Dual active learning for reinforcement learning from human feedback (2024). https:\/\/doi.org\/10.48550\/arXiv.2410.02504","DOI":"10.48550\/arXiv.2410.02504"},{"key":"45_CR8","doi-asserted-by":"publisher","unstructured":"Liu, X., et\u00a0al.: Agentbench: evaluating LLMs as agents. arXiv preprint arXiv:2308.03688 (2023). https:\/\/doi.org\/10.48550\/arXiv.2308.03688","DOI":"10.48550\/arXiv.2308.03688"},{"key":"45_CR9","doi-asserted-by":"publisher","unstructured":"Liu, Z., et al.: Understanding R1-zero-like training: a critical perspective (2025). https:\/\/doi.org\/10.48550\/arXiv.2503.20783","DOI":"10.48550\/arXiv.2503.20783"},{"key":"45_CR10","unstructured":"Luo, M., et al.: DeepSWE: training a fully open-sourced, state-of-the-art coding agent by scaling RL. https:\/\/www.together.ai\/blog\/deepswe"},{"key":"45_CR11","doi-asserted-by":"publisher","unstructured":"Qin, Y., et al.: ToolLLM: facilitating large language models to master 16000+ real-world APIs (2023). https:\/\/doi.org\/10.48550\/arXiv.2307.16789","DOI":"10.48550\/arXiv.2307.16789"},{"key":"45_CR12","doi-asserted-by":"publisher","unstructured":"Qwen, Yang, A., et al.: Qwen2.5 Technical Report (2025). https:\/\/doi.org\/10.48550\/arXiv.2412.15115","DOI":"10.48550\/arXiv.2412.15115"},{"key":"45_CR13","doi-asserted-by":"crossref","unstructured":"Rouzegar, H., Makrehchi, M.: Enhancing text classification through LLM-driven active learning and human annotation. In: Henning, S., Stede, M. (eds.) Proceedings of the 18th Linguistic Annotation Workshop (LAW-XVIII), pp. 98\u2013111. Association for Computational Linguistics, St. Julians (2024)","DOI":"10.18653\/v1\/2024.law-1.10"},{"key":"45_CR14","doi-asserted-by":"publisher","unstructured":"Shao, Z., et al.: DeepSeekMath: pushing the limits of mathematical reasoning in open language models (2024). https:\/\/doi.org\/10.48550\/arXiv.2402.03300","DOI":"10.48550\/arXiv.2402.03300"},{"key":"45_CR15","doi-asserted-by":"publisher","unstructured":"Shridhar, M., Yuan, X., C\u00f4t\u00e9, M.A., Bisk, Y., Trischler, A., Hausknecht, M.J.: ALFWorld: aligning text and embodied environments for interactive learning. In: 9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, 3\u20137 May 2021. OpenReview.net (2021). https:\/\/doi.org\/10.48550\/arXiv.2010.03768","DOI":"10.48550\/arXiv.2010.03768"},{"key":"45_CR16","doi-asserted-by":"publisher","unstructured":"Thil, L.A., Popa, M., Spanakis, G.: Navigating WebAI: training agents to complete web tasks with large language models and reinforcement learning. In: Proceedings of the 39th ACM\/SIGAPP Symposium on Applied Computing, pp. 866\u2013874 (2024). https:\/\/doi.org\/10.1145\/3605098.3635903","DOI":"10.1145\/3605098.3635903"},{"key":"45_CR17","doi-asserted-by":"publisher","unstructured":"Trivedi, H., et al.: AppWorld: a controllable world of apps and people for benchmarking interactive coding agents (2024). https:\/\/doi.org\/10.48550\/arXiv.2407.18901","DOI":"10.48550\/arXiv.2407.18901"},{"key":"45_CR18","doi-asserted-by":"publisher","unstructured":"Wang, Z., et al.: RAGEN: understanding self-evolution in LLM agents via multi-turn reinforcement learning (2025). https:\/\/doi.org\/10.48550\/arXiv.2504.20073","DOI":"10.48550\/arXiv.2504.20073"},{"key":"45_CR19","doi-asserted-by":"publisher","unstructured":"Xie, T., et\u00a0al.: Osworld: benchmarking multimodal agents for open-ended tasks in real computer environments. Adv. Neural Inf. Process. Syst. 37, 52040\u201352094 (2024). https:\/\/doi.org\/10.48550\/arXiv.2404.07972","DOI":"10.48550\/arXiv.2404.07972"},{"key":"45_CR20","doi-asserted-by":"publisher","unstructured":"Xu, Y., et al.: MobileRL: online agentic reinforcement learning for mobile GUI agents (2025). https:\/\/doi.org\/10.48550\/arXiv.2509.18119","DOI":"10.48550\/arXiv.2509.18119"},{"key":"45_CR21","doi-asserted-by":"publisher","unstructured":"Xue, Z., et al.: SimpleTIR: end-to-end reinforcement learning for multi-turn tool-integrated reasoning (2025). https:\/\/doi.org\/10.48550\/arXiv.2509.02479","DOI":"10.48550\/arXiv.2509.02479"},{"key":"45_CR22","doi-asserted-by":"publisher","unstructured":"Yao, S., Chen, H., Yang, J., Narasimhan, K.: WebShop: towards scalable real-world web interaction with grounded language agents. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, 28 November\u20139 December 2022 (2022). https:\/\/doi.org\/10.48550\/arXiv.2207.01206","DOI":"10.48550\/arXiv.2207.01206"},{"key":"45_CR23","doi-asserted-by":"publisher","unstructured":"Yao, S., Shinn, N., Razavi, P., Narasimhan, K.: $$\\tau $$-bench: a benchmark for tool-agent-user interaction in real-world domains. CoRR arxiv:2406.12045 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2406.12045","DOI":"10.48550\/ARXIV.2406.12045"},{"key":"45_CR24","doi-asserted-by":"publisher","unstructured":"Yu, Q., et al.: DAPO: an open-source LLM reinforcement learning system at scale (2025). https:\/\/doi.org\/10.48550\/arXiv.2503.14476","DOI":"10.48550\/arXiv.2503.14476"},{"key":"45_CR25","doi-asserted-by":"publisher","unstructured":"Yue, Y., et al.: Does reinforcement learning really incentivize reasoning capacity in LLMs beyond the base model? (2025). https:\/\/doi.org\/10.48550\/arXiv.2504.13837","DOI":"10.48550\/arXiv.2504.13837"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2856-0_45","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:18:50Z","timestamp":1783783130000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2856-0_45"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,12]]},"ISBN":["9789819228553","9789819228560"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2856-0_45","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,12]]},"assertion":[{"value":"12 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2026.rosc.org.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}