{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,30]],"date-time":"2025-12-30T07:01:20Z","timestamp":1767078080674,"version":"3.48.0"},"reference-count":15,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Control Syst. Lett."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/lcsys.2025.3642767","type":"journal-article","created":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T18:34:45Z","timestamp":1765391685000},"page":"2879-2884","source":"Crossref","is-referenced-by-count":0,"title":["Training Task Reasoning LLM Agents for Multi-Turn Task Planning via Single-Turn Reinforcement Learning"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5698-5887","authenticated-orcid":false,"given":"Hanjiang","family":"Hu","sequence":"first","affiliation":[{"name":"Mitsubishi Electric Research Laboratories, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3767-5517","authenticated-orcid":false,"given":"Changliu","family":"Liu","sequence":"additional","affiliation":[{"name":"Robotics Institute, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9545-3050","authenticated-orcid":false,"given":"Na","family":"Li","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7209-7866","authenticated-orcid":false,"given":"Yebin","family":"Wang","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Large language model agent: A survey on methodology, applications and challenges","author":"Luo","year":"2025","journal-title":"arXiv:2503.21460"},{"key":"ref2","first-page":"1","article-title":"React: Synergizing reasoning and acting in language models","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Yao"},{"key":"ref3","article-title":"Evaluating LLM-based agents for multi-turn conversations: A survey","author":"Guan","year":"2025","journal-title":"arXiv:2503.22458"},{"key":"ref4","article-title":"RAGEN: Understanding self-evolution in LLM agents via multi-turn reinforcement learning","author":"Wang","year":"2025","journal-title":"arXiv:2504.20073"},{"key":"ref5","article-title":"Group-in-group policy optimization for LLM agent training","author":"Feng","year":"2025","journal-title":"arXiv:2505.10978"},{"key":"ref6","first-page":"5244","article-title":"AGILE: A novel reinforcement learning framework of LLM agents","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"37","author":"Peiyuan"},{"key":"ref7","article-title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models","author":"Shao","year":"2024","journal-title":"arXiv:2402.03300"},{"key":"ref8","article-title":"Reinforcement learning with verifiable rewards: GRPO\u2019s effective loss, dynamics, and success amplification","author":"Mroueh","year":"2025","journal-title":"arXiv:2503.06639"},{"key":"ref9","first-page":"1","article-title":"Robotouille: An asynchronous planning benchmark for LLM agents","volume-title":"Proc. 13th Int. Conf. Learn. Represent.","author":"Gonzalez-Pumariega"},{"key":"ref10","article-title":"The Llama 3 herd of models","author":"Dubey","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3689031.3696075"},{"key":"ref12","article-title":"Qwen2.5 technical report","volume-title":"arXiv:2412.15115","author":"Yang","year":"2024"},{"key":"ref13","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.662"},{"key":"ref15","article-title":"Reinforce++: An efficient RLHF algorithm with robustness to both prompt and reward models","author":"Hu","year":"2025","journal-title":"arXiv:2501.03262"}],"container-title":["IEEE Control Systems Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7782633\/10939047\/11293775.pdf?arnumber=11293775","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,30]],"date-time":"2025-12-30T06:59:13Z","timestamp":1767077953000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11293775\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":15,"URL":"https:\/\/doi.org\/10.1109\/lcsys.2025.3642767","relation":{},"ISSN":["2475-1456"],"issn-type":[{"type":"electronic","value":"2475-1456"}],"subject":[],"published":{"date-parts":[[2025]]}}}