{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T14:34:50Z","timestamp":1784644490716,"version":"3.55.0"},"reference-count":32,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Science and Technology Major Project of China","award":["2022ZD0120002"],"award-info":[{"award-number":["2022ZD0120002"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62233004"],"award-info":[{"award-number":["62233004"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62303112"],"award-info":[{"award-number":["62303112"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jiangsu Provincial Scientific Research Center of Applied Mathematics","award":["BK20233002"],"award-info":[{"award-number":["BK20233002"]}]},{"DOI":"10.13039\/501100004608","name":"Natural Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["BK20230826"],"award-info":[{"award-number":["BK20230826"]}],"id":[{"id":"10.13039\/501100004608","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1109\/tac.2025.3570065","type":"journal-article","created":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T13:35:25Z","timestamp":1747229725000},"page":"7109-7124","source":"Crossref","is-referenced-by-count":11,"title":["Distributed Neural Policy Gradient Algorithm for Global Convergence of Networked Multiagent Reinforcement Learning"],"prefix":"10.1109","volume":"70","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1455-7216","authenticated-orcid":false,"given":"Pengcheng","family":"Dai","sequence":"first","affiliation":[{"name":"School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0500-5201","authenticated-orcid":false,"given":"Yuanqiu","family":"Mo","sequence":"additional","affiliation":[{"name":"School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0301-9180","authenticated-orcid":false,"given":"Wenwu","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Mathematics, Frontiers Science Center for Mobile Information Communication and Security, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2818-9752","authenticated-orcid":false,"given":"Wei","family":"Ren","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of California, Riverside, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2019.2933443"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3082639"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2901791"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.3015811"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2016.12.020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2020.3018871"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2019.2933973"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2019.2933417"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220096"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"ref12","first-page":"6846","article-title":"Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rashid","year":"2018"},{"key":"ref13","article-title":"QPLEX: Duplex dueling multi-agent Q-learning","author":"Wang","year":"2020"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref15","article-title":"Off-policy multi-agent decomposed policy gradients","author":"Wang","year":"2020"},{"key":"ref16","first-page":"5872","article-title":"Fully decentralized multi-agent reinforcement learning with networked agents","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang","year":"2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3139138"},{"key":"ref18","article-title":"Neural policy gradient methods: Global optimality and rates of convergence","author":"Wang","year":"2019"},{"key":"ref19","first-page":"1626","article-title":"Finite-time analysis of distributed TD(0) with linear function approximation on multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Doan","year":"2019"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9029257"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2020.12.2021"},{"key":"ref22","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"12","author":"Sutton","year":"1999"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2023.1370"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2007.11.026"},{"key":"ref25","first-page":"3101","article-title":"Optimistic policy iteration and natural actor-critic: A unifying view and a non-optimality result","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"26","author":"Wagner","year":"2013"},{"key":"ref26","first-page":"4358","article-title":"Improving sample complexity bounds for (natural) actor-critic algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Xu","year":"2020"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tac.2025.3570065"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2010.2041686"},{"key":"ref29","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-319-91578-4","volume-title":"Introductory Lectures on Convex Optimization","author":"Nesterov","year":"2018"},{"key":"ref30","first-page":"2563","article-title":"Convergence rates for localized actor-critic in networked Markov potential games","volume-title":"Proc. Uncertainty Artif. Intell. Conf.","author":"Zhou","year":"2023"},{"key":"ref31","first-page":"6820","article-title":"On the global convergence rates of softmax policy gradient methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mei","year":"2020"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2018.2817461"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9\/11218261\/11003569.pdf?arnumber=11003569","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,23]],"date-time":"2025-12-23T06:16:50Z","timestamp":1766470610000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11003569\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":32,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tac.2025.3570065","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"value":"0018-9286","type":"print"},{"value":"1558-2523","type":"electronic"},{"value":"2334-3303","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11]]}}}