{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T16:54:33Z","timestamp":1783184073794,"version":"3.54.6"},"reference-count":133,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/CAA J. Autom. Sinica"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/jas.2026.125993","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T19:49:09Z","timestamp":1777578549000},"page":"776-795","source":"Crossref","is-referenced-by-count":1,"title":["Understanding Agentic AI: Algorithms and Infrastructure"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6305-1740","authenticated-orcid":false,"given":"Wanlun","family":"Ma","sequence":"first","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9279-3010","authenticated-orcid":false,"given":"Yongjian","family":"Guo","sequence":"additional","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7207-0716","authenticated-orcid":false,"given":"Qing-Long","family":"Han","sequence":"additional","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2548-6348","authenticated-orcid":false,"given":"Wei","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0647-4747","authenticated-orcid":false,"given":"Xiaogang","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Information Technology, Adelaide University,Adelaide,South Australia,Australia,5005"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2028-510X","authenticated-orcid":false,"given":"Junwu","family":"Xiong","sequence":"additional","affiliation":[{"name":"AI Infra Team at JD Technology,Beijing,China,100176"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0655-666X","authenticated-orcid":false,"given":"Sheng","family":"Wen","sequence":"additional","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5252-0831","authenticated-orcid":false,"given":"Yang","family":"Xiang","sequence":"additional","affiliation":[{"name":"School of Science, Computing and Emerging Technologies, Swinburne University of Technology,Hawthorn,VIC,Australia,3122"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Describe, explain, plan and select: Interactive planning with large language models enables open-world multitask agents","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Wang"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/s44336-024-00009-2"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3586183.3606763"},{"key":"ref4","article-title":"Ghost in the Minecraft: Generally capable agents for open-world environments via large language models with text-based knowledge and memory","author":"Zhu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref5","article-title":"Voyager: An openended embodied agent with large language models","author":"Wang","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-024-40678-2"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2997"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.810"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2025.125552"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124983"},{"key":"ref11","article-title":"Plan and budget: Effective and efficient test-time scaling on reasoning large language models","author":"Lin","year":"2025","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"A survey on test-time scaling in large language models: What, how, where, and how well?","author":"Zhang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref13","article-title":"StreamRL: Scalable, heterogeneous, and elastic RL for LLMs with disaggregated stream generation","author":"Zhong","year":"2025","journal-title":"arXiv preprint"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3689031.3696075"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3767295.3803580"},{"key":"ref16","article-title":"Reinforcement learning optimization for large-scale learning: An efficient and user-friendly scaling library","author":"Wang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00713"},{"key":"ref18","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv preprint"},{"key":"ref19","article-title":"IMPACT: Importance weighted asynchronous architectures with clipped target networks","volume-title":"Proc. 8th Int. Conf. Learning Representations","author":"Luo"},{"key":"ref20","article-title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models","author":"Shao","year":"2024","journal-title":"arXiv preprint"},{"key":"ref21","article-title":"DAPO: An open-source LLM reinforcement learning system at scale","author":"Yu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref22","article-title":"Asymmetric proximal policy optimization: Mini-critics boost LLM reasoning","author":"Liu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.464"},{"key":"ref24","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Rafailov"},{"key":"ref25","first-page":"4447","article-title":"A general theoretical paradigm to understand learning from human preferences","volume-title":"Proc. 27th Int. Conf. Artificial Intelligence and Statistics","author":"Azar"},{"key":"ref26","first-page":"12634","article-title":"Model alignment as prospect theoretic optimization","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Ethayarajh"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0482"},{"key":"ref28","article-title":"SLiC-HF: Sequence likelihood calibration with human feedback","author":"Zhao","year":"2023","journal-title":"arXiv preprint"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.626"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3946"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1802"},{"key":"ref32","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Proc. 36th Int. Conf. Neural Information Processing Systems","author":"Wei"},{"key":"ref33","article-title":"Self-consistency improves chain of thought reasoning in language models","volume-title":"Proc. Eleventh Int. Conf. Learning Representations","author":"Wang"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2019"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0517"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-75538-8_7"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2066"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29720"},{"key":"ref39","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021","journal-title":"arXiv preprint"},{"key":"ref40","article-title":"LEVER: Learning to verify language-to-code generation with execution","volume-title":"Proc. 40th Int. Conf. Machine Learning","author":"Ni"},{"key":"ref41","article-title":"Let\u2019s verify step by step","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Lightman"},{"key":"ref42","article-title":"Generative verifiers: Reward modeling as next-token prediction","volume-title":"Proc. Thirteenth Int. Conf. Learning Representations","author":"Zhang"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.167"},{"key":"ref44","article-title":"When to solve, when to verify: Compute-optimal problem solving and generative verification for LLM reasoning","volume-title":"Proc. Second Conf. Language Modeling","author":"Singhi"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1269"},{"key":"ref46","article-title":"Large language model cascades with mixture of thought representations for cost-efficient reasoning","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Yue"},{"key":"ref47","article-title":"Skeleton-of-thought: Prompting LLMs for efficient parallel generation","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Ning"},{"key":"ref48","article-title":"DeepSpeed-chat: Easy, fast and affordable RLHF training of ChatGPT-like models at all scales","author":"Yao","year":"2023","journal-title":"arXiv preprint"},{"key":"ref49","volume-title":"TRL: Transformer reinforcement learning","author":"Von Werra","year":"2020"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i28.35383"},{"key":"ref51","volume-title":"slime: An LLM post-training framework for RL scaling","author":"Zhu","year":"2025"},{"key":"ref52","article-title":"RAGEN: Understanding self-evolution in LLM agents via multi-turn reinforcement learning","author":"Wang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref53","article-title":"Optimizing RLHF training for large language models with stage fusion","volume-title":"Proc. 22nd USENIX Symp. on Networked Systems Design and Implementation","author":"Zhong"},{"key":"ref54","article-title":"DistFlow: A fully distributed RL framework for scalable and efficient LLM post-training","author":"Wang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref55","article-title":"OpenRLHF: An easy-to-use, scalable and high-performance RLHF framework","author":"Hu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref56","volume-title":"NeMo RL: A scalable and efficient post-training library","year":"2025"},{"key":"ref57","volume-title":"SkyRL-v0: Train real-world long-horizon agents via reinforcement learning","author":"Cao","year":"2025"},{"key":"ref58","article-title":"AReal: A large-scale asynchronous reinforcement learning system for language reasoning","author":"Fu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref59","article-title":"SeamlessFlow: A trainer agent isolation RL framework achieving bubble-free pipelines via tag scheduling","author":"Wang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref60","article-title":"AsyncFlow: An asynchronous streaming RL framework for efficient LLM post-training","author":"Han","year":"2025","journal-title":"arXiv preprint"},{"key":"ref61","article-title":"AWorld: Orchestrating the training recipe for agentic AI","author":"Yu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref62","volume-title":"Verifiers: Environments for LLM reinforcement learning","author":"Brown","year":"2025"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref64","article-title":"SWE-bench: Can language models resolve real-world GitHub issues?","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Jimenez"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2025.125498"},{"key":"ref66","article-title":"AgentGym-RL: Training LLM agents for long-horizon decision making through multi-turn reinforcement learning","author":"Xi","year":"2025","journal-title":"arXiv preprint"},{"key":"ref67","article-title":"Kimi k1.5: Scaling reinforcement learning with LLMs","author":"Team","year":"2025","journal-title":"arXiv preprint"},{"key":"ref68","article-title":"LongCat-flash-Omni technical report","author":"Wang","year":"2025","journal-title":"arXiv preprint"},{"key":"ref69","article-title":"React: Synergizing reasoning and acting in language models","volume-title":"Proc. Eleventh Int. Conf. Learning Representations","author":"Yao"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0377"},{"key":"ref71","article-title":"Self-RAG: Learning to retrieve, generate, and critique through self-reflection","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Asai"},{"key":"ref72","article-title":"ToolAlpaca: Generalized tool learning for language models with 3000 simulated cases","author":"Tang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref73","article-title":"ToolLLM: Facilitating large language models to master 16000+ real-world APIs","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Qin"},{"key":"ref74","article-title":"Gorilla: Large language model connected with massive APIs","volume-title":"Proc. 38th Int. Conf. Neural Information Processing Systems","author":"Patil"},{"key":"ref75","article-title":"Chameleon: Plug-and-play compositional reasoning with large language models","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Lu"},{"key":"ref76","first-page":"2727","article-title":"Efficient tool use with chain-of-abstraction reasoning","volume-title":"Proc. 31st Int. Conf. Computational Linguistics","author":"Gao"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3382"},{"key":"ref78","article-title":"Identifying the risks of lm agents with an lm-emulated sandbox","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Ruan"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.44"},{"key":"ref80","article-title":"HuggingGPT: Solving AI tasks with ChatGPT and its friends in Hugging Face","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Shen"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1387"},{"key":"ref82","article-title":"Language agent tree search unifies reasoning, acting, and planning in language models","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Zhou"},{"key":"ref83","article-title":"LLM+P: Empowering large language models with optimal planning proficiency","author":"Liu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref84","article-title":"Plan-and-Act: Improving planning of agents for long-horizon tasks","volume-title":"Proc. Forty-Second Int. Conf. Machine Learning","author":"Erdogan"},{"key":"ref85","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume-title":"Proc. 39th Int. Conf. Machine Learning","author":"Huang"},{"key":"ref86","first-page":"1769","article-title":"Inner monologue: Embodied reasoning through planning with language models","volume-title":"Proc. 6th Conf. Robot Learning","author":"Huang"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00280"},{"key":"ref88","article-title":"Agent planning with world knowledge model","volume-title":"Proc. 38th Int. Conf. Neural Information Processing Systems","author":"Qiao"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.747"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29946"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3259"},{"key":"ref92","article-title":"MemGPT: Towards LLMs as operating systems","author":"Packer","year":"2023","journal-title":"arXiv preprint"},{"key":"ref93","article-title":"A human-inspired reading agent with gist memory of very long contexts","volume-title":"Proc. 41st Int. Conf. Machine Learning","author":"Lee","year":"2024"},{"key":"ref94","article-title":"A-MEM: Agentic memory for LLM agents","volume-title":"Proc. 39th Conf. Neural Information Processing Systems","author":"Xu"},{"key":"ref95","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","volume-title":"Proc. 34th Int. Conf. Neural Information Processing Systems","author":"Lewis"},{"issue":"251","key":"ref96","first-page":"1","article-title":"Atlas: Few-shot learning with retrieval augmented language models","volume":"24","author":"Izacard","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1145\/3777378"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124671"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124743"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124365"},{"key":"ref101","article-title":"CAMEL: Communicative agents for \u2018mind\u2019 exploration of large language model society","volume-title":"Proc. 37th Int. Conf. Neural Information Processing Systems","author":"Li","year":"2023"},{"key":"ref102","article-title":"AgentVerse: Facilitating multi-agent collaboration and exploring emergent behaviors","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Chen"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.15"},{"key":"ref104","article-title":"MetaGPT: Meta programming for a multi-agent collaborative framework","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Hong"},{"key":"ref105","article-title":"Fleet of agents: Coordinated problem solving with large language models","volume-title":"Proc. Forty-Second Int. Conf. Machine Learning","author":"Klein"},{"key":"ref106","article-title":"ChatEval: Towards better LLM-based evaluators through multi-agent debate","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Chan"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.992"},{"key":"ref108","volume-title":"CrewAI","year":"2025"},{"key":"ref109","article-title":"GameGPT: Multi-agent collaborative framework for game development","author":"Chen","year":"2023","journal-title":"arXiv preprint"},{"key":"ref110","volume-title":"Autogen","year":"2023"},{"key":"ref111","volume-title":"Swarm","year":"2024"},{"key":"ref112","volume-title":"Agentstack","year":"2025"},{"key":"ref113","volume-title":"Strands agents SDK for Python","year":"2025"},{"key":"ref114","volume-title":"Langchain: The platform for reliable agents","year":"2023"},{"key":"ref115","volume-title":"Semantic kernel","year":"2025"},{"key":"ref116","volume-title":"LlamaIndex","year":"2025"},{"key":"ref117","volume-title":"LangGraph: Build resilient language agents as graphs","year":"2024"},{"key":"ref118","volume-title":"LangFlow","year":"2024"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2025.125327"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2024.124407"},{"key":"ref121","article-title":"AnyPos: Automated task-agnostic actions for bimanual manipulation","author":"Tan","year":"2025","journal-title":"arXiv preprint"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.015"},{"key":"ref123","article-title":"Vidar: Embodied video diffusion model for generalist manipulation","author":"Feng","year":"2025","journal-title":"arXiv preprint"},{"key":"ref124","first-page":"2679","article-title":"OpenVLA: An open-source vision-language-action model","volume-title":"Proc. 8th Conf. Robot Learning","author":"Kim"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2025.xxi.010"},{"key":"ref126","article-title":"GROOT N1: An open foundation model for generalist humanoid robots","author":"Bjorck","year":"2025","journal-title":"arXiv preprint"},{"key":"ref127","volume-title":"LeRobot: State-of-the-art machine learning for real-world robotics in Pytorch","author":"Cadene","year":"2024"},{"key":"ref128","article-title":"RLinf: Flexible and efficient large-scale reinforcement learning via macro-to-micro flow transformation","author":"Yu","year":"2025","journal-title":"arXiv preprint"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1939"},{"key":"ref130","article-title":"Visual imitation enables contextual humanoid control","author":"Allshire","year":"2025","journal-title":"arXiv preprint"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/TFUZZ.2025.3567089"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.114109"},{"key":"ref133","article-title":"WebArena: A realistic web environment for building autonomous agents","volume-title":"Proc. Twelfth Int. Conf. Learning Representations","author":"Zhou"}],"container-title":["IEEE\/CAA Journal of Automatica Sinica"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6570654\/11503144\/11503205.pdf?arnumber=11503205","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T19:55:24Z","timestamp":1777665324000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11503205\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":133,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/jas.2026.125993","relation":{},"ISSN":["2329-9266","2329-9274"],"issn-type":[{"value":"2329-9266","type":"print"},{"value":"2329-9274","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}