{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:45:53Z","timestamp":1782798353853,"version":"3.54.5"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,18]],"date-time":"2026-05-18T00:00:00Z","timestamp":1779062400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,18]]},"DOI":"10.1109\/infocom59046.2026.11571537","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:38:15Z","timestamp":1782761895000},"page":"1-10","source":"Crossref","is-referenced-by-count":0,"title":["Global Convergence for Multi-agent Reinforcement Learning in Unreliable Communication Networks"],"prefix":"10.1109","author":[{"given":"Pengcheng","family":"Dai","sequence":"first","affiliation":[{"name":"Singapore University of Technology and Design,Pillar of Engineering Systems and Design,Singapore,487372"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingjie","family":"Duan","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology (Guangzhou),Internet of Things Thrust,Guangzhou,China,511453"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2022.3141105"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1080\/08839514.2019.1600301"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2025.3541078"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2016.12.020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2020.3018871"},{"issue":"1","key":"ref9","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref10","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Duan"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2019.2933443"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3082639"},{"key":"ref13","article-title":"Multi-agent reinforcement learning for networked system control","author":"Chu","year":"2020"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.3015811"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2019.2933973"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2019.2933417"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220096"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ton.2025.3647225"},{"key":"ref20","first-page":"2074","article-title":"Scalable multi-agent reinforcement learning for networked systems with average reward","volume-title":"Proc. Adv. Neural Inf. Process. Syst","author":"Qu"},{"key":"ref21","first-page":"7825","article-title":"Multi-agent reinforcement learning in stochastic networked systems","volume-title":"Proc. Adv. Neural Inf. Process. Syst","author":"Lin"},{"key":"ref22","first-page":"256","article-title":"Scalable reinforcement learning of localized policies for multi-agent networked systems","volume-title":"Proc. Learn. Dyn. Control","author":"Qu"},{"key":"ref23","first-page":"6820","article-title":"On the global convergence rates of softmax policy gradient methods","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Mei"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0140"},{"key":"ref25","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst","author":"Sutton"},{"key":"ref26","first-page":"2563","article-title":"Convergence rates for localized actor-critic in networked markov potential games","volume-title":"Proc. Uncertainty Artif. Intell","author":"Zhou"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2025.3570065"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/9.580874"}],"event":{"name":"IEEE INFOCOM 2026 - IEEE Conference on Computer Communications","location":"Tokyo, Japan","start":{"date-parts":[[2026,5,18]]},"end":{"date-parts":[[2026,5,21]]}},"container-title":["IEEE INFOCOM 2026 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11571071\/11571169\/11571537.pdf?arnumber=11571537","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:30:24Z","timestamp":1782797424000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11571537\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,18]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/infocom59046.2026.11571537","relation":{},"subject":[],"published":{"date-parts":[[2026,5,18]]}}}