{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T16:34:17Z","timestamp":1783701257831,"version":"3.55.0"},"reference-count":34,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Automatica"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.automatica.2026.113008","type":"journal-article","created":{"date-parts":[[2026,4,19]],"date-time":"2026-04-19T15:51:34Z","timestamp":1776613894000},"page":"113008","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Reinforcement learning for output consensus of multi-agent systems under hierarchical control"],"prefix":"10.1016","volume":"189","author":[{"given":"Yingying","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongzhi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaodi","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhanshan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"12","key":"10.1016\/j.automatica.2026.113008_b1","doi-asserted-by":"crossref","first-page":"8753","DOI":"10.1016\/j.jfranklin.2022.01.012","article-title":"Model-free distributed optimal consensus control of nonlinear multi-agent systems: A graphical game approach","volume":"360","author":"An","year":"2023","journal-title":"Journal of the Franklin Institute"},{"issue":"9","key":"10.1016\/j.automatica.2026.113008_b2","doi-asserted-by":"crossref","first-page":"5570","DOI":"10.1002\/rnc.7280","article-title":"Model-free distributed optimal control for general discrete-time linear systems using reinforcement learning","volume":"34","author":"Feng","year":"2024","journal-title":"International Journal of Robust and Nonlinear Control"},{"key":"10.1016\/j.automatica.2026.113008_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2026.112833","article-title":"Zonotopic state estimation and fault detection for discrete-time linear systems under DoS attacks: A switching controller design mechanism","volume":"185","author":"Guo","year":"2026","journal-title":"Automatica"},{"issue":"1","key":"10.1016\/j.automatica.2026.113008_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.jfranklin.2025.108259","article-title":"Using time delayed disturbance compensation for sliding mode control","volume":"363","author":"Han","year":"2026","journal-title":"Journal of the Franklin Institute"},{"issue":"11","key":"10.1016\/j.automatica.2026.113008_b5","doi-asserted-by":"crossref","first-page":"1443","DOI":"10.1049\/cth2.12473","article-title":"Optimal consensus control for multi-agent systems: Multi-step policy gradient adaptive dynamic programming method","volume":"17","author":"Ji","year":"2023","journal-title":"IET Control Theory & Applications"},{"key":"10.1016\/j.automatica.2026.113008_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2020.109149","article-title":"Cooperative adaptive optimal output regulation of nonlinear discrete-time multi-agent systems","volume":"121","author":"Jiang","year":"2020","journal-title":"Automatica"},{"key":"10.1016\/j.automatica.2026.113008_b7","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2022.110768","article-title":"Reinforcement learning and cooperative H\u221e output regulation of linear continuous-time multi-agent systems","volume":"148","author":"Jiang","year":"2023","journal-title":"Automatica"},{"issue":"1","key":"10.1016\/j.automatica.2026.113008_b8","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1049\/iet-cta.2012.0855","article-title":"Leaderless and leader-following consensus for heterogeneous multi-agent systems with random link failures","volume":"8","author":"Kim","year":"2014","journal-title":"IET Control Theory & Applications"},{"issue":"4","key":"10.1016\/j.automatica.2026.113008_b9","doi-asserted-by":"crossref","first-page":"1167","DOI":"10.1016\/j.automatica.2014.02.015","article-title":"Reinforcement Q-learning for optimal tracking control of linear discrete-time systems with unknown dynamics","volume":"50","author":"Kiumarsi","year":"2014","journal-title":"Automatica"},{"key":"10.1016\/j.automatica.2026.113008_b10","series-title":"Optimal control","author":"Lewis","year":"2012"},{"key":"10.1016\/j.automatica.2026.113008_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2021.110076","article-title":"Off-policy Q-learning: Solving Nash equilibrium of multi-player games with network-induced delay and unmeasured state","volume":"136","author":"Li","year":"2022","journal-title":"Automatica"},{"key":"10.1016\/j.automatica.2026.113008_b12","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.neucom.2022.10.032","article-title":"Optimal consensus of a class of discrete-time linear multi-agent systems via value iteration with guaranteed admissibility","volume":"516","author":"Li","year":"2023","journal-title":"Neurocomputing"},{"key":"10.1016\/j.automatica.2026.113008_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.111430","article-title":"A novel data-driven model-free synchronization protocol for discrete-time multi-agent systems via TD3 based algorithm","volume":"287","author":"Liu","year":"2024","journal-title":"Knowledge-Based Systems"},{"issue":"6","key":"10.1016\/j.automatica.2026.113008_b14","doi-asserted-by":"crossref","first-page":"3468","DOI":"10.1109\/TCYB.2023.3276797","article-title":"Data-driven optimal bipartite consensus control for second-order multiagent systems via policy gradient reinforcement learning","volume":"54","author":"Liu","year":"2023","journal-title":"IEEE Transactions on Cybernetics"},{"issue":"6","key":"10.1016\/j.automatica.2026.113008_b15","doi-asserted-by":"crossref","first-page":"3666","DOI":"10.1109\/TSMC.2022.3230504","article-title":"Data-driven bipartite consensus tracking for nonlinear multiagent systems with prescribed performance","volume":"53","author":"Liu","year":"2023","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"issue":"14","key":"10.1016\/j.automatica.2026.113008_b16","doi-asserted-by":"crossref","first-page":"10564","DOI":"10.1016\/j.jfranklin.2023.08.010","article-title":"Model-free algorithm for consensus of discrete-time multi-agent systems using reinforcement learning method","volume":"360","author":"Long","year":"2023","journal-title":"Journal of the Franklin Institute"},{"issue":"11","key":"10.1016\/j.automatica.2026.113008_b17","doi-asserted-by":"crossref","first-page":"3051","DOI":"10.1109\/TAC.2014.2317301","article-title":"Linear quadratic tracking control of partially-unknown continuous-time systems using reinforcement learning","volume":"59","author":"Modares","year":"2014","journal-title":"IEEE Transactions on Automatic Control"},{"issue":"12","key":"10.1016\/j.automatica.2026.113008_b18","doi-asserted-by":"crossref","first-page":"10951","DOI":"10.1109\/TIE.2019.2958277","article-title":"Optimal model-free output synchronization of heterogeneous multiagent systems under switching topologies","volume":"67","author":"Mu","year":"2020","journal-title":"IEEE Transactions on Industrial Electronics"},{"issue":"4","key":"10.1016\/j.automatica.2026.113008_b19","doi-asserted-by":"crossref","first-page":"7379","DOI":"10.1109\/TASE.2023.3341801","article-title":"Adaptive RL optimized bipartite consensus tracking for heterogeneous nonlinear MASs under a switching threshold event triggered strategy","volume":"21","author":"Niu","year":"2024","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"issue":"9","key":"10.1016\/j.automatica.2026.113008_b20","doi-asserted-by":"crossref","first-page":"3689","DOI":"10.1109\/TCSI.2022.3177407","article-title":"Distributed optimal tracking control of discrete-time multiagent systems via event-triggered reinforcement learning","volume":"69","author":"Peng","year":"2022","journal-title":"IEEE Transactions on Circuits and Systems. I. Regular Papers"},{"issue":"1","key":"10.1016\/j.automatica.2026.113008_b21","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TIE.2023.3247734","article-title":"Hybrid iteration ADP algorithm to solve cooperative, optimal output regulation problem for continuous-time, linear, multiagent systems: Theory and application in islanded modern microgrids with IBRs","volume":"71","author":"Qasem","year":"2023","journal-title":"IEEE Transactions on Industrial Electronics"},{"issue":"12","key":"10.1016\/j.automatica.2026.113008_b22","doi-asserted-by":"crossref","first-page":"7933","DOI":"10.1109\/TCYB.2023.3240983","article-title":"Cooperative differential game-based distributed optimal synchronization control of heterogeneous nonlinear multiagent systems","volume":"53","author":"Sun","year":"2023","journal-title":"IEEE Transactions on Cybernetics"},{"issue":"9","key":"10.1016\/j.automatica.2026.113008_b23","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-022-3683-y","article-title":"Event-triggered consensus control of heterogeneous multi-agent systems: Model-and data-based approaches","volume":"66","author":"Wang","year":"2023","journal-title":"Science China. Information Sciences"},{"key":"10.1016\/j.automatica.2026.113008_b24","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1109\/TSIPN.2023.3239654","article-title":"Cooperative learning of multi-agent systems via reinforcement learning","volume":"9","author":"Wang","year":"2023","journal-title":"IEEE Transactions on Signal and Information Processing over Networks"},{"issue":"1","key":"10.1016\/j.automatica.2026.113008_b25","doi-asserted-by":"crossref","DOI":"10.1155\/2023\/6350647","article-title":"Model-free cooperative optimal output regulation for linear discrete-time multi-agent systems using reinforcement learning","volume":"2023","author":"Wu","year":"2023","journal-title":"Mathematical Problems in Engineering"},{"issue":"11","key":"10.1016\/j.automatica.2026.113008_b26","doi-asserted-by":"crossref","first-page":"15984","DOI":"10.1109\/TNNLS.2023.3291542","article-title":"Data-based optimal synchronization of heterogeneous multiagent systems in graphical games via reinforcement learning","volume":"35","author":"Xiong","year":"2023","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.automatica.2026.113008_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.110221","article-title":"Design of data-driven mode-free iterative learning controller based higher order parameter estimation for multi-agent systems consistency tracking","volume":"261","author":"Xu","year":"2023","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.automatica.2026.113008_b28","first-page":"1","article-title":"Adaptive critic learning-based optimal bipartite consensus for multiagent systems with prescribed performance","author":"Yan","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"3","key":"10.1016\/j.automatica.2026.113008_b29","doi-asserted-by":"crossref","first-page":"651","DOI":"10.1016\/j.jfranklin.2012.12.015","article-title":"Distributed event-triggered control of discrete-time heterogeneous multi-agent systems","volume":"350","author":"Yin","year":"2013","journal-title":"Journal of the Franklin Institute"},{"issue":"3","key":"10.1016\/j.automatica.2026.113008_b30","doi-asserted-by":"crossref","first-page":"651","DOI":"10.1016\/j.jfranklin.2012.12.015","article-title":"Distributed event-triggered control of discrete-time heterogeneous multi-agent systems","volume":"350","author":"Yin","year":"2013","journal-title":"Journal of the Franklin Institute"},{"issue":"11","key":"10.1016\/j.automatica.2026.113008_b31","doi-asserted-by":"crossref","first-page":"6953","DOI":"10.1109\/TSMC.2023.3289796","article-title":"Data-driven distributed adaptive consensus tracking of nonlinear multiagent systems: A controller-based dynamic linearization method","volume":"53","author":"Yu","year":"2023","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"issue":"8","key":"10.1016\/j.automatica.2026.113008_b32","doi-asserted-by":"crossref","first-page":"3651","DOI":"10.1109\/TCYB.2025.3568813","article-title":"Dynamic event-triggered nonsingular predefined-time tracking control for fully heterogeneous vehicle platoon with spacing constraints","volume":"55","author":"Zhang","year":"2025","journal-title":"IEEE Transactions on Cybernetics"},{"issue":"7","key":"10.1016\/j.automatica.2026.113008_b33","doi-asserted-by":"crossref","first-page":"4661","DOI":"10.1016\/j.jfranklin.2023.02.036","article-title":"Data-driven control of consensus tracking for discrete-time multi-agent systems","volume":"360","author":"Zhang","year":"2023","journal-title":"Journal of the Franklin Institute"},{"key":"10.1016\/j.automatica.2026.113008_b34","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2023.111198","article-title":"Consensus tracking control for a class of general linear hybrid multi-agent systems: A model-free approach","volume":"156","author":"Zhou","year":"2023","journal-title":"Automatica"}],"container-title":["Automatica"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0005109826001925?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0005109826001925?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T10:58:33Z","timestamp":1779447513000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0005109826001925"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":34,"alternative-id":["S0005109826001925"],"URL":"https:\/\/doi.org\/10.1016\/j.automatica.2026.113008","relation":{},"ISSN":["0005-1098"],"issn-type":[{"value":"0005-1098","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Reinforcement learning for output consensus of multi-agent systems under hierarchical control","name":"articletitle","label":"Article Title"},{"value":"Automatica","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.automatica.2026.113008","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113008"}}