{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,22]],"date-time":"2026-02-22T07:00:49Z","timestamp":1771743649496,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":34,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819570805","type":"print"},{"value":"9789819570812","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-7081-2_44","type":"book-chapter","created":{"date-parts":[[2026,2,22]],"date-time":"2026-02-22T06:45:02Z","timestamp":1771742702000},"page":"639-651","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PEARL: Plan Exploration and\u00a0Adaptive Reinforcement Learning for\u00a0Multihop Tool Use"],"prefix":"10.1007","author":[{"given":"Qihao","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingzhe","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiayue","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yue","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanbing","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,23]]},"reference":[{"key":"44_CR1","unstructured":"Anthropic: Introducing Claude 3.5 sonnet. https:\/\/www.anthropic.com\/news\/claude-3-5-sonnet (2024)"},{"key":"44_CR2","unstructured":"Besta, M., et\u00a0al.: Graph of thoughts: Solving elaborate problems with large language models. arXiv preprint arXiv:2308.09687 (2023)"},{"key":"44_CR3","unstructured":"Chen, Z., et\u00a0al.: T-Eval: Evaluating the tool utilization capability of large language models step by step. arXiv preprint arXiv:2312.14033 (2023)"},{"key":"44_CR4","unstructured":"Gur, I., et al.: A real-world WebAgent with planning, long context understanding, and program synthesis. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"44_CR5","unstructured":"Huang, J., et al.: An embodied generalist agent in 3D world. arXiv preprint arXiv:2311.12871 (2023)"},{"key":"44_CR6","doi-asserted-by":"crossref","unstructured":"Kong, Y., et\u00a0al.: TPTU-v2: Boosting task planning and tool usage of large language model-based agents in real-world systems. arXiv preprint arXiv:2311.11315 (2023)","DOI":"10.18653\/v1\/2024.emnlp-industry.27"},{"key":"44_CR7","doi-asserted-by":"crossref","unstructured":"Li, M., Song, F., Yu, B., Yu, H., Li, Z., Huang, F., Li, Y.: API-bank: A benchmark for tool-augmented LLMs. arXiv preprint arXiv:2304.08244 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.187"},{"key":"44_CR8","unstructured":"Liu, J., et al.: RLTF: Reinforcement learning from unit test feedback. arXiv preprint arXiv:2307.04349 (2023)"},{"key":"44_CR9","unstructured":"Liu, W., et\u00a0al.: ToolACE: Winning the points of LLM function calling. arXiv preprint arXiv:2409.00920 (2024)"},{"key":"44_CR10","unstructured":"Liu, Y., et al.: Tool-planner: Task planning with clusters across multiple tools. arXiv preprint arXiv:2406.03807 (2024)"},{"key":"44_CR11","unstructured":"OpenAI: OpenAI: Introducing ChatGPT (2022). https:\/\/openai.com\/blog\/chatgpt"},{"key":"44_CR12","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems. vol.\u00a035, pp. 27730\u201327744. Curran Associates, Inc. (2022)"},{"key":"44_CR13","unstructured":"Parisi, A., Zhao, Y., Fiedel, N.: TALM: Tool augmented language models. arXiv preprint arXiv:2205.12255 (2022)"},{"key":"44_CR14","unstructured":"Patil, S.G., Zhang, T., Wang, X., Gonzalez, J.E.: Gorilla: Large language model connected with massive APIs. arXiv preprint arXiv:2305.15334 (2023)"},{"key":"44_CR15","unstructured":"Qian, C., et al.: ToolRL: Reward is all tool learning needs. arXiv preprint arXiv:2504.13958 (2025)"},{"key":"44_CR16","doi-asserted-by":"crossref","unstructured":"Qian, C., Xiong, C., Liu, Z., Liu, Z.: Toolink: linking toolkit creation and using through chain-of-solving on open-source model. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 831\u2013854 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.48"},{"key":"44_CR17","unstructured":"Qiao, S., Gui, H., Chen, H., Zhang, N.: Making language models better tool learners with execution feedback. arXiv preprint arXiv:2305.13068 (2023)"},{"key":"44_CR18","first-page":"68539","volume":"36","author":"T Schick","year":"2023","unstructured":"Schick, T., et al.: Toolformer: language models can teach themselves to use tools. Adv. Neural. Inf. Process. Syst. 36, 68539\u201368551 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"44_CR19","unstructured":"Shao, Z., et\u00a0al.: DeepSeekMath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)"},{"key":"44_CR20","unstructured":"Shen, Y., Song, K., Tan, X., Li, D., Lu, W., Zhuang, Y.: HuggingGPT: Solving ai tasks with ChatGPT and its friends in HuggingFace. arXiv preprint arXiv:2303.17580 (2023)"},{"key":"44_CR21","doi-asserted-by":"crossref","unstructured":"Shi, Z., et al.: Tool learning in the wild: empowering language models as automatic tool agents. In: Proceedings of the ACM on Web Conference 2025, pp. 2222\u20132237 (2025)","DOI":"10.1145\/3696410.3714825"},{"key":"44_CR22","doi-asserted-by":"crossref","unstructured":"Sun, J., Min, S.Y., Chang, Y., Bisk, Y.: Tools fail: Detecting silent errors in faulty tools. arXiv preprint arXiv:2406.19228 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.790"},{"key":"44_CR23","unstructured":"Team, Q.: Qwen2.5: A party of foundation models (2024). https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"44_CR24","unstructured":"Wang, G., et al.: Voyager: an open-ended embodied agent with large language models. Transactions on Machine Learning Research (2024)"},{"key":"44_CR25","unstructured":"Wang, L., et\u00a0al.: A survey on large language model based autonomous agents. arXiv preprint arXiv:2308.11432 (2023)"},{"key":"44_CR26","unstructured":"Wang, Z., Cai, S., Liu, A., Ma, X., Liang, Y.: Describe, explain, plan and select: Interactive planning with large language models enables open-world multi-task agents. arXiv preprint arXiv:2302.01560 (2023)"},{"key":"44_CR27","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural. Inf. Process. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"44_CR28","unstructured":"Wu, C., Yin, S., Qi, W., Wang, X., Tang, Z., Duan, N.: Visual ChatGPT: Talking, drawing and editing with visual foundation models. arXiv preprint arXiv:2303.04671 (2023)"},{"key":"44_CR29","doi-asserted-by":"crossref","unstructured":"Wu, Q., Liu, W., Luan, J., Wang, B.: ToolPlanner: A tool augmented LLM for multi granularity instructions with path planning and feedback. arXiv preprint arXiv:2409.14826 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.1018"},{"key":"44_CR30","first-page":"11809","volume":"36","author":"S Yao","year":"2023","unstructured":"Yao, S., et al.: Tree of thoughts: deliberate problem solving with large language models. Adv. Neural. Inf. Process. Syst. 36, 11809\u201311822 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"44_CR31","unstructured":"Yao, S., et al.: ReAct: Synergizing reasoning and acting in language models. arXiv preprint arXiv:2210.03629 (2022)"},{"key":"44_CR32","doi-asserted-by":"crossref","unstructured":"Ye, J., et\u00a0al.: ToolHop: A query-driven benchmark for evaluating large language models in multi-hop tool use. arXiv preprint arXiv:2501.02506 (2025)","DOI":"10.18653\/v1\/2025.acl-long.150"},{"key":"44_CR33","unstructured":"Yu, Y., et al.: StepTool: A step-grained reinforcement learning framework for tool learning in LLMs. arXiv preprint arXiv:2410.07745 (2024)"},{"key":"44_CR34","unstructured":"Yuan, H., Yuan, Z., Tan, C., Wang, W., Huang, S., Huang, F.: RRHF: rank responses to align language models with human feedback. In: Thirty-seventh Conference on Neural Information Processing Systems (2023)"}],"container-title":["Lecture Notes in Computer Science","PRICAI 2025: Trends in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-7081-2_44","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,22]],"date-time":"2026-02-22T06:45:15Z","timestamp":1771742715000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-7081-2_44"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819570805","9789819570812"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-7081-2_44","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"23 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRICAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific Rim International Conference on Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Wellington","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Zealand","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pricai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.pricai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}