{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T21:26:23Z","timestamp":1740173183803,"version":"3.37.3"},"reference-count":29,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"23","license":[{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,12,1]],"date-time":"2023-12-01T00:00:00Z","timestamp":1701388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022ZD0120002"],"award-info":[{"award-number":["2022ZD0120002"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62233004","62073076"],"award-info":[{"award-number":["62233004","62073076"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004608","name":"Natural Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["BK20210216"],"award-info":[{"award-number":["BK20210216"]}],"id":[{"id":"10.13039\/501100004608","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100019059","name":"Jiangsu Provincial Key Laboratory of Networked Collective Intelligence","doi-asserted-by":"publisher","award":["BM2017002"],"award-info":[{"award-number":["BM2017002"]}],"id":[{"id":"10.13039\/501100019059","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Internet Things J."],"published-print":{"date-parts":[[2023,12,1]]},"DOI":"10.1109\/jiot.2023.3284510","type":"journal-article","created":{"date-parts":[[2023,6,9]],"date-time":"2023-06-09T17:27:09Z","timestamp":1686331629000},"page":"21039-21060","source":"Crossref","is-referenced-by-count":0,"title":["Distributed Neural Learning Algorithms for Multiagent Reinforcement Learning"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1455-7216","authenticated-orcid":false,"given":"Pengcheng","family":"Dai","sequence":"first","affiliation":[{"name":"Jiangsu Key Laboratory of Networked Collective Intelligence, School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6887-368X","authenticated-orcid":false,"given":"Hongzhe","family":"Liu","sequence":"additional","affiliation":[{"name":"Jiangsu Key Laboratory of Networked Collective Intelligence, School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6406-2399","authenticated-orcid":false,"given":"Wenwu","family":"Yu","sequence":"additional","affiliation":[{"name":"Frontiers Science Center for Mobile Information Communication and Security, School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"He","family":"Wang","sequence":"additional","affiliation":[{"name":"Jiangsu Key Laboratory of Networked Collective Intelligence, School of Mathematics, Southeast University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2020.2986803"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/jiot.2022.3187067"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2901791"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.3015811"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2019.2933443"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3082639"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2016.12.020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2020.3018871"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2019.2933973"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2019.2933417"},{"key":"ref14","article-title":"Towards characterizing divergence in deep Q-learning","author":"Achiam","year":"2019","journal-title":"arXiv:1903.08894"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/9.580874"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2020.2024"},{"key":"ref17","first-page":"4026","article-title":"Stochastic variance-reduced policy gradient","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Papini"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1137\/19M1288012"},{"key":"ref19","first-page":"541","article-title":"An improved convergence analysis of stochastic variance-reduced policy gradient","volume-title":"Proc. Uncertainty Artif. Intell. Conf.","author":"Xu"},{"key":"ref20","first-page":"1008","article-title":"Actor\u2013critic algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"12","author":"Konda"},{"key":"ref21","article-title":"On the global convergence of actor\u2013critic: A case for linear quadratic regulator with ergodic cost","author":"Yang","year":"2019","journal-title":"arXiv:1907.06246"},{"key":"ref22","first-page":"11315","article-title":"Neural temporal-difference learning converges to global optima","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cai"},{"key":"ref23","first-page":"1","article-title":"Neural policy gradient methods: Global optimality and rates of convergence","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220096"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2013.2241057"},{"key":"ref26","first-page":"1626","article-title":"Finite-time analysis of distributed TD(0) with linear function approximation on multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Doan"},{"key":"ref27","first-page":"9340","article-title":"Fully decentralized multi-agent reinforcement learning with networked agents","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2021.3139138"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2010.2041686"}],"container-title":["IEEE Internet of Things Journal"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6488907\/10323344\/10147339.pdf?arnumber=10147339","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,11]],"date-time":"2024-04-11T04:40:10Z","timestamp":1712810410000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10147339\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,1]]},"references-count":29,"journal-issue":{"issue":"23"},"URL":"https:\/\/doi.org\/10.1109\/jiot.2023.3284510","relation":{},"ISSN":["2327-4662","2372-2541"],"issn-type":[{"type":"electronic","value":"2327-4662"},{"type":"electronic","value":"2372-2541"}],"subject":[],"published":{"date-parts":[[2023,12,1]]}}}