{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T16:07:50Z","timestamp":1784563670571,"version":"3.55.0"},"reference-count":40,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.engappai.2026.115538","type":"journal-article","created":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T12:27:02Z","timestamp":1783340822000},"page":"115538","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"P4","title":["Progressive subgoal-aggregated long-sequence decision-making with large language models"],"prefix":"10.1016","volume":"181","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0412-5858","authenticated-orcid":false,"given":"Yadong","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiliang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8292-6389","authenticated-orcid":false,"given":"Zhen","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2026.115538_b1","series-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022"},{"key":"10.1016\/j.engappai.2026.115538_b2","series-title":"Hindsight experience replay","author":"Andrychowicz","year":"2017"},{"key":"10.1016\/j.engappai.2026.115538_b3","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"The option-critic architecture","volume":"vol. 31","author":"Bacon","year":"2017"},{"key":"10.1016\/j.engappai.2026.115538_b4","series-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019"},{"key":"10.1016\/j.engappai.2026.115538_b5","series-title":"Learning with AMIGo: Adversarially motivated intrinsic goals","author":"Campero","year":"2021"},{"issue":"3209","key":"10.1016\/j.engappai.2026.115538_b6","first-page":"73383","article-title":"Minigrid & miniworld: Modular & customizable reinforcement learning environments for goal-oriented tasks","volume":"36","author":"Chevalier-Boisvert","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115538_b7","series-title":"Advances in Neural Information Processing Systems","first-page":"13049","article-title":"Emergent complexity and zero-shot transfer via unsupervised environment design","volume":"vol. 33","author":"Dennis","year":"2020"},{"key":"10.1016\/j.engappai.2026.115538_b8","series-title":"Proceedings of the 40th International Conference on Machine Learning","first-page":"8657","article-title":"Guiding pretraining in reinforcement learning with large language models","volume":"vol. 202","author":"Du","year":"2023"},{"key":"10.1016\/j.engappai.2026.115538_b9","series-title":"Large language model based multi-agents: A survey of progress and challenges","author":"Guo","year":"2024"},{"key":"10.1016\/j.engappai.2026.115538_b10","doi-asserted-by":"crossref","first-page":"37631","DOI":"10.52202\/068431-2728","article-title":"Exploration via elliptical episodic bonuses","volume":"35","author":"Henaff","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.engappai.2026.115538_b11","series-title":"Enabling intelligent interactions between an agent and an LLM: A reinforcement learning approach","author":"Hu","year":"2023"},{"key":"10.1016\/j.engappai.2026.115538_b12","series-title":"Proceedings of the 39th International Conference on Machine Learning","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume":"vol. 162","author":"Huang","year":"2022"},{"key":"10.1016\/j.engappai.2026.115538_b13","series-title":"The NetHack learning environment","first-page":"7671","author":"K\u00fcttler","year":"2020"},{"key":"10.1016\/j.engappai.2026.115538_b14","series-title":"Hierarchical reinforcement learning with hindsight","author":"Levy","year":"2018"},{"key":"10.1016\/j.engappai.2026.115538_b15","series-title":"Goal-conditioned reinforcement learning: Problems and solutions","author":"Liu","year":"2022"},{"key":"10.1016\/j.engappai.2026.115538_b16","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume":"48","author":"Mnih","year":"2016","journal-title":"Proc. the 33rd Int. Conf. Mach. Learn."},{"issue":"7540","key":"10.1016\/j.engappai.2026.115538_b17","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"Mnih","year":"2015","journal-title":"Nature"},{"issue":"2","key":"10.1016\/j.engappai.2026.115538_b18","doi-asserted-by":"crossref","first-page":"265","DOI":"10.1109\/TEVC.2006.890271","article-title":"Intrinsic motivation systems for autonomous mental development","volume":"11","author":"Oudeyer","year":"2007","journal-title":"IEEE Trans. Evol. Comput."},{"issue":"5","key":"10.1016\/j.engappai.2026.115538_b19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3453160","article-title":"Hierarchical reinforcement learning: A comprehensive survey","volume":"54","author":"Pateria","year":"2021","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.engappai.2026.115538_b20","series-title":"International Conference on Machine Learning","first-page":"2778","article-title":"Curiosity-driven exploration by self-supervised prediction","volume":"vol. 70","author":"Pathak","year":"2017"},{"key":"10.1016\/j.engappai.2026.115538_b21","series-title":"LLM augmented hierarchical agents","author":"Prakash","year":"2023"},{"key":"10.1016\/j.engappai.2026.115538_b22","series-title":"Sentence embeddings using siamese BERT-networks","author":"Reimers","year":"2019"},{"key":"10.1016\/j.engappai.2026.115538_b23","series-title":"Minihack the planet: A sandbox for open-ended reinforcement learning research","author":"Samvelyan","year":"2021"},{"key":"10.1016\/j.engappai.2026.115538_b24","series-title":"Proceedings of the 32nd International Conference on Machine Learning","first-page":"1312","article-title":"Universal value function approximators","volume":"vol. 37","author":"Schaul","year":"2015"},{"issue":"3","key":"10.1016\/j.engappai.2026.115538_b25","doi-asserted-by":"crossref","first-page":"230","DOI":"10.1109\/TAMD.2010.2056368","article-title":"Formal theory of creativity, fun, and intrinsic motivation (1990\u20132010)","volume":"2","author":"Schmidhuber","year":"2010","journal-title":"IEEE Trans. Auton. Ment. Dev."},{"key":"10.1016\/j.engappai.2026.115538_b26","series-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"10.1016\/j.engappai.2026.115538_b27","series-title":"Reflexion: language agents with verbal reinforcement learning","first-page":"8634","author":"Shinn","year":"2023"},{"issue":"7587","key":"10.1016\/j.engappai.2026.115538_b28","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of go with deep neural networks and tree search","volume":"529","author":"Silver","year":"2016","journal-title":"Nature"},{"issue":"7676","key":"10.1016\/j.engappai.2026.115538_b29","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of go without human knowledge","volume":"550","author":"Silver","year":"2017","journal-title":"Nature"},{"key":"10.1016\/j.engappai.2026.115538_b30","series-title":"Reinforcement learning: An introduction","author":"Sutton","year":"2018"},{"key":"10.1016\/j.engappai.2026.115538_b31","series-title":"International Conference on Machine Learning","first-page":"3540","article-title":"Feudal networks for hierarchical reinforcement learning","author":"Vezhnevets","year":"2017"},{"issue":"7782","key":"10.1016\/j.engappai.2026.115538_b32","doi-asserted-by":"crossref","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","article-title":"Grandmaster level in StarCraft II using multi-agent reinforcement learning","volume":"575","author":"Vinyals","year":"2019","journal-title":"Nature"},{"key":"10.1016\/j.engappai.2026.115538_b33","series-title":"DEIR: efficient and robust exploration through discriminative-model-based episodic intrinsic rewards","author":"Wan","year":"2023"},{"issue":"6","key":"10.1016\/j.engappai.2026.115538_b34","doi-asserted-by":"crossref","DOI":"10.1007\/s11704-024-40231-1","article-title":"A survey on large language model based autonomous agents","volume":"18","author":"Wang","year":"2024","journal-title":"Front. Comput. Sci."},{"key":"10.1016\/j.engappai.2026.115538_b35","series-title":"Voyager: An open-ended embodied agent with large language models","author":"Wang","year":"2023"},{"issue":"2","key":"10.1016\/j.engappai.2026.115538_b36","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-024-4222-0","article-title":"The rise and potential of large language model based agents: A survey","volume":"68","author":"Xi","year":"2025","journal-title":"Sci. China Inf. Sci."},{"key":"10.1016\/j.engappai.2026.115538_b37","series-title":"Foundation models for decision making: Problems, methods, and opportunities","author":"Yang","year":"2023"},{"key":"10.1016\/j.engappai.2026.115538_b38","series-title":"React: Synergizing reasoning and acting in language models","author":"Yao","year":"2022"},{"key":"10.1016\/j.engappai.2026.115538_b39","series-title":"The World Wide Web Conference","first-page":"3620","article-title":"CityFlow: A multi-agent reinforcement learning environment for large scale city traffic scenario","author":"Zhang","year":"2019"},{"key":"10.1016\/j.engappai.2026.115538_b40","series-title":"Large language model as a policy teacher for training reinforcement learning agents","author":"Zhou","year":"2023"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626018221?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197626018221?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T15:45:50Z","timestamp":1784562350000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197626018221"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":40,"alternative-id":["S0952197626018221"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115538","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Progressive subgoal-aggregated long-sequence decision-making with large language models","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2026.115538","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115538"}}