{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T05:01:37Z","timestamp":1784869297108,"version":"3.55.0"},"reference-count":48,"publisher":"Elsevier BV","issue":"13","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of the Franklin Institute"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.jfranklin.2026.108817","type":"journal-article","created":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:28:50Z","timestamp":1782948530000},"page":"108817","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-agent reinforcement learning control with Lyapunov stability guarantees for cooperative systems"],"prefix":"10.1016","volume":"363","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0563-5796","authenticated-orcid":false,"given":"Chao","family":"Jia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiarui","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jfranklin.2026.108817_bib0001","series-title":"IEEE\/RSJ International Conference on Intelligent Robots and Systems","first-page":"3694","article-title":"Multi-robot box-pushing: single-agent Q-learning vs. team Q-learning","author":"Wang","year":"2006"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0002","first-page":"1","article-title":"An overview of multi-agent reinforcement learning from game theoretical perspective","volume":"abs\/2011.00583","author":"Yang","year":"2020","journal-title":"CoRR"},{"issue":"9","key":"10.1016\/j.jfranklin.2026.108817_bib0003","doi-asserted-by":"crossref","first-page":"2419","DOI":"10.1007\/s10994-021-05961-4","article-title":"Challenges of real-world reinforcement learning: definitions, benchmarks and analysis","volume":"110","author":"Dulac-Arnold","year":"2021","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0004","series-title":"Proceedings of the Eleventh International Conference on Machine Learning","first-page":"157","article-title":"Markov games as a framework for multi-agent reinforcement learning","author":"Littman","year":"1994"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0005","first-page":"1039","article-title":"Nash Q-learning for general-sum stochastic games","volume":"4","author":"Hu","year":"2003","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0006","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"6382","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","author":"Lowe","year":"2017"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0007","series-title":"Proceedings of the 35th International Conference on Machine Learning (ICML)","first-page":"4295","article-title":"QMIX: monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"vol. 80","author":"Rashid","year":"2018"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0008","series-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks","article-title":"Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks","author":"Papoudakis","year":"2021"},{"issue":"6","key":"10.1016\/j.jfranklin.2026.108817_bib0009","doi-asserted-by":"crossref","first-page":"750","DOI":"10.1007\/s10458-019-09421-1","article-title":"A survey and critique of multiagent deep reinforcement learning","volume":"33","author":"Hernandez-Leal","year":"2019","journal-title":"Auton. Agents Multi-Agent Syst."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0010","series-title":"Markov Chains and Stochastic Stability","author":"Meyn","year":"1993"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0011","series-title":"Nonlinear Systems: Analysis, Stability, and Control","author":"Sastry","year":"1999"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0012","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"A Lyapunov-based approach to safe reinforcement learning","volume":"vol. 31","author":"Chow","year":"2018"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0013","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"908","article-title":"Safe model-based reinforcement learning with stability guarantees","author":"Berkenkamp","year":"2017"},{"issue":"2","key":"10.1016\/j.jfranklin.2026.108817_bib0014","doi-asserted-by":"crossref","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","article-title":"A comprehensive survey of multiagent reinforcement learning","volume":"38","author":"Busoniu","year":"2008","journal-title":"IEEE Trans. Syst. Man Cybern. C"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0015","series-title":"Handbook of Reinforcement Learning and Control","first-page":"321","article-title":"Multi-agent reinforcement learning: a selective overview of theories and algorithms","author":"Zhang","year":"2021"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0016","first-page":"1","article-title":"String stable control of connected vehicles via multi-agent Lyapunov actor-critic","volume":"DXqdLSvG26","author":"Fu","year":"2025","journal-title":"OpenReview"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0017","unstructured":"C. Zhang, L. Wei, J. Fan, Z. Liu, Y. Huang, Lyapunov-guided multi-agent reinforcement learning for delay-sensitive wireless scheduling, 2024, arXiv: 2411.01766."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0018","series-title":"Proceedings of the 35th International Conference on Machine Learning","first-page":"1861","article-title":"Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume":"vol. 80","author":"Haarnoja","year":"2018"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0019","doi-asserted-by":"crossref","DOI":"10.1016\/j.ins.2025.122514","article-title":"Sequence value decomposition transformer for cooperative multi-agent reinforcement learning","volume":"720","author":"Zhao","year":"2025","journal-title":"Inf. Sci."},{"issue":"7","key":"10.1016\/j.jfranklin.2026.108817_bib0020","doi-asserted-by":"crossref","first-page":"3794","DOI":"10.3390\/app15073794","article-title":"QMIX-GNN: a graph neural network-based heterogeneous multi-agent reinforcement learning model for improved collaboration and decision-making","volume":"15","author":"Zhao","year":"2025","journal-title":"Appl. Sci."},{"issue":"7","key":"10.1016\/j.jfranklin.2026.108817_bib0021","doi-asserted-by":"crossref","first-page":"12521","DOI":"10.1109\/TNNLS.2024.3455422","article-title":"TVDO: Tchebycheff value-decomposition optimization for multiagent reinforcement learning","volume":"36","author":"Hu","year":"2025","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0022","first-page":"1","article-title":"CoMIX: a multi-agent reinforcement learning training architecture for efficient decentralized coordination and independent decision-making","volume":"JoU9khOwwr","author":"Minelli","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0023","unstructured":"D. Bolliger, L. Zauter, R. Ziegler, Fully-decentralized MADDPG with networked agents, 2025, arXiv: 2503.06747."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0024","series-title":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"1","article-title":"Multi-agent hierarchical graph attention actor-critic reinforcement learning","author":"Li","year":"2025"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0025","series-title":"Proceedings of the 34th International Conference on Machine Learning (ICML)","first-page":"22","article-title":"Constrained policy optimization","volume":"vol. 70","author":"Achiam","year":"2017"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0026","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"3387","article-title":"End-to-end safe reinforcement learning through barrier functions for safety-critical continuous control tasks","volume":"vol. 33","author":"Cheng","year":"2019"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0027","first-page":"1","article-title":"MAS3AC: a learning framework for general multi-agent safe and stable control with state-wise guarantees","volume":"gjD9i0sOhx","author":"Zhao","year":"2026","journal-title":"OpenReview"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0028","unstructured":"N.A. Zeyang Li, Safe multi-agent reinforcement learning with convergence to generalized Nash equilibrium, 2024, arXiv: 2411.15036."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0029","unstructured":"Y. Zhang, Y. Xing, Q. Quan, Z. She, Multi-step actor-critic learning with Lyapunov certificates for exponentially stabilizing control, 2025, arXiv: 2512.24955."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0030","series-title":"Advances in Neural Information Processing Systems","first-page":"20646","article-title":"Certifying stability of reinforcement learning policies using generalized Lyapunov functions","volume":"vol. 38","author":"Long","year":"2025"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0031","doi-asserted-by":"crossref","first-page":"289","DOI":"10.1613\/jair.2447","article-title":"Optimal and approximate Q-value functions for decentralized POMDPs","volume":"32","author":"Oliehoek","year":"2008","journal-title":"J. Artif. Intell. Res."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0032","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1016\/j.neucom.2016.01.031","article-title":"Multi-agent reinforcement learning as a rehearsal for decentralized planning","volume":"190","author":"Kraemer","year":"2016","journal-title":"Neurocomputing"},{"issue":"2","key":"10.1016\/j.jfranklin.2026.108817_bib0033","doi-asserted-by":"crossref","first-page":"408","DOI":"10.1109\/TPAMI.2013.218","article-title":"Gaussian processes for data-efficient learning in robotics and control","volume":"37","author":"Deisenroth","year":"2015","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0034","series-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"1998"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0035","series-title":"Proceedings of 1995 34th IEEE Conference on Decision and Control","first-page":"560","article-title":"Neuro-dynamic programming: an overview","volume":"vol. 1","author":"Bertsekas","year":"1995"},{"issue":"1","key":"10.1016\/j.jfranklin.2026.108817_bib0036","doi-asserted-by":"crossref","first-page":"8","DOI":"10.1073\/pnas.53.1.8","article-title":"On the stability of stochastic dynamical systems","volume":"53","author":"Kushner","year":"1965","journal-title":"Proc. Natl. Acad. Sci."},{"issue":"1","key":"10.1016\/j.jfranklin.2026.108817_bib0037","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1109\/JPROC.2006.887293","article-title":"Consensus and cooperation in networked multi-agent systems","volume":"95","author":"Olfati-Saber","year":"2007","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0038","series-title":"Proceedings of the 23rd National Conference on Artificial Intelligence","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"vol. 3","author":"Ziebart","year":"2008"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0039","series-title":"Proceedings of the 35th International Conference on Machine Learning","first-page":"5872","article-title":"Fully decentralized multi-agent reinforcement learning with networked agents","volume":"vol. 80","author":"Zhang","year":"2018"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0040","series-title":"Stochastic Stability of Differential Equations","author":"Khasminskii","year":"2012"},{"key":"10.1016\/j.jfranklin.2026.108817_bib0041","series-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"Puterman","year":"1994"},{"issue":"5","key":"10.1016\/j.jfranklin.2026.108817_bib0042","doi-asserted-by":"crossref","first-page":"359","DOI":"10.1016\/0893-6080(89)90020-8","article-title":"Multilayer feedforward networks are universal approximators","volume":"2","author":"Hornik","year":"1989","journal-title":"Neural Netw."},{"issue":"4","key":"10.1016\/j.jfranklin.2026.108817_bib0043","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/BF02551274","article-title":"Approximation by superpositions of a sigmoidal function","volume":"2","author":"Cybenko","year":"1989","journal-title":"Math. Control Signals Syst."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0044","doi-asserted-by":"crossref","DOI":"10.1016\/j.automatica.2023.110999","article-title":"Deep reinforcement learning control approach to mitigating actuator attacks","volume":"152","author":"Wu","year":"2023","journal-title":"Automatica"},{"issue":"8","key":"10.1016\/j.jfranklin.2026.108817_bib0045","doi-asserted-by":"crossref","first-page":"1871","DOI":"10.1360\/SSI-2024-0050","article-title":"Multi-UAV collaborative path planning based on multi-agent soft actor critic (in Chinese)","volume":"54","author":"Fang","year":"2024","journal-title":"Sci Sin Inf."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0046","series-title":"UAV-Path-Planning","year":"2024"},{"issue":"3","key":"10.1016\/j.jfranklin.2026.108817_bib0047","doi-asserted-by":"crossref","first-page":"400","DOI":"10.1214\/aoms\/1177729586","article-title":"A stochastic approximation method","volume":"22","author":"Robbins","year":"1951","journal-title":"Ann. Math. Stat."},{"key":"10.1016\/j.jfranklin.2026.108817_bib0048","series-title":"On the Sample Complexity of Reinforcement Learning","author":"Kakade","year":"2003"}],"container-title":["Journal of the Franklin Institute"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0016003226004163?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0016003226004163?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T04:18:47Z","timestamp":1784866727000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0016003226004163"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":48,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["S0016003226004163"],"URL":"https:\/\/doi.org\/10.1016\/j.jfranklin.2026.108817","relation":{},"ISSN":["0016-0032"],"issn-type":[{"value":"0016-0032","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Multi-agent reinforcement learning control with Lyapunov stability guarantees for cooperative systems","name":"articletitle","label":"Article Title"},{"value":"Journal of the Franklin Institute","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jfranklin.2026.108817","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Franklin Institute. Published by Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108817"}}