{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T17:31:53Z","timestamp":1781717513373,"version":"3.54.5"},"reference-count":58,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071425"],"award-info":[{"award-number":["62071425"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Key Research and Development Plan","award":["2022C01093"],"award-info":[{"award-number":["2022C01093"]}]},{"name":"Huawei Cooperation Project"},{"DOI":"10.13039\/501100004731","name":"Zhejiang Provincial Natural Science Foundation of China","doi-asserted-by":"publisher","award":["LR23F010005"],"award-info":[{"award-number":["LR23F010005"]}],"id":[{"id":"10.13039\/501100004731","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Wireless Commun."],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1109\/twc.2023.3279268","type":"journal-article","created":{"date-parts":[[2023,6,9]],"date-time":"2023-06-09T17:29:18Z","timestamp":1686331758000},"page":"507-528","source":"Crossref","is-referenced-by-count":24,"title":["The Gradient Convergence Bound of Federated Multi-Agent Reinforcement Learning With Efficient Communication"],"prefix":"10.1109","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4564-470X","authenticated-orcid":false,"given":"Xing","family":"Xu","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4297-5060","authenticated-orcid":false,"given":"Rongpeng","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5479-7890","authenticated-orcid":false,"given":"Zhifeng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang Lab, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1492-1364","authenticated-orcid":false,"given":"Honggang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang Lab, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1093\/comjnl\/bxz129"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TCE.2017.015014"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2018.03.008"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-32259-7_13"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/IISA.2015.7387990"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref7","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","volume":"37","author":"Schulman"},{"key":"ref8","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref9","article-title":"Tsallis reinforcement learning: A unified framework for maximum entropy reinforcement learning","author":"Lee","year":"2019","journal-title":"arXiv:1902.00137"},{"issue":"1","key":"ref10","first-page":"237","article-title":"Reinforcement learning: A survey","volume":"4","author":"Littman","year":"1996","journal-title":"J. Artif. Intell. Res."},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/b978-1-55860-307-3.50049-6"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3056418"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155494"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MIS.2020.2994942"},{"key":"ref17","article-title":"Parle: Parallelizing stochastic gradient descent","author":"Chaudhari","year":"2017","journal-title":"arXiv:1707.00424"},{"key":"ref18","first-page":"1","article-title":"CoCoA: A general framework for communication-efficient distributed optimization","volume":"18","author":"Smith","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015693"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2006.887293"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2022.3143495"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2022.3216326"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2022.3183465"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2023.3242734"},{"key":"ref25","article-title":"Towards flexible device participation in federated learning","author":"Ruan","year":"2020","journal-title":"arXiv:2006.06954"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295340"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/SPAWC48557.2020.9154332"},{"key":"ref28","article-title":"Asynchronous decentralized parallel stochastic gradient descent","author":"Lian","year":"2017","journal-title":"arXiv:1710.06952"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/MOBHOC.2007.4428658"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCC.2007.913919"},{"key":"ref32","article-title":"Asynchronous methods for deep reinforcement learning","author":"Mnih","year":"2016","journal-title":"arXiv:1602.01783"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-05816-6_3"},{"key":"ref34","article-title":"Federated transfer reinforcement learning for autonomous driving","author":"Liang","year":"2019","journal-title":"arXiv:1910.06001"},{"key":"ref35","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"Proc. 31st Int. Conf. Mach. Learn.","author":"Silver"},{"key":"ref36","article-title":"Soft actor-critic algorithms and applications","author":"Haarnoja","year":"2018","journal-title":"arXiv:1812.05905"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2020.2986803"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref39","article-title":"Concentrated differentially private and utility preserving federated learning","author":"Hu","year":"2020","journal-title":"arXiv:2003.13761"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1561\/2200000083"},{"key":"ref41","article-title":"Federated learning: Strategies for improving communication efficiency","author":"Konecn\u00fd","year":"2016","journal-title":"arXiv:1610.05492"},{"key":"ref42","article-title":"Cooperative SGD: A unified framework for the design and analysis of communication-efficient SGD algorithms","author":"Wang","year":"2018","journal-title":"arXiv:1808.07576"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8486403"},{"key":"ref44","article-title":"Local SGD with periodic averaging: Tighter analysis and adaptive synchronization","author":"Haddadpour","year":"2019","journal-title":"arXiv:1910.13598"},{"key":"ref45","first-page":"5325","article-title":"Error compensated quantized SGD and its applications to large-scale distributed optimization","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","volume":"80","author":"Wu"},{"key":"ref46","first-page":"9850","article-title":"ATOMO: Communication-efficient learning via atomic sparsification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Wang"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM41043.2020.9155269"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TETC.2020.3043300"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/LCOMM.2022.3144451"},{"key":"ref50","first-page":"1246","article-title":"Gradient descent only converges to minimizers","volume-title":"Proc. 29th Annu. Conf. Learn. Theory","volume":"49","author":"Lee"},{"key":"ref51","first-page":"2681","article-title":"Deep decentralized multi-task multi-agent reinforcement learning under partial observability","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Omidshafiei"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/203330.203343"},{"key":"ref53","article-title":"Differentially private federated learning for resource-constrained Internet of Things","author":"Hu","year":"2020","journal-title":"arXiv:2003.12705"},{"key":"ref54","first-page":"2737","article-title":"Asynchronous parallel stochastic gradient for nonconvex optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Lian"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1137\/16M1080173"},{"key":"ref56","article-title":"On the Fenchel duality between strong convexity and Lipschitz continuous gradient","author":"Zhou","year":"2018","journal-title":"arXiv:1803.06573"},{"key":"ref57","first-page":"399","article-title":"Benchmarks for reinforcement learning in mixed-autonomy traffic","volume-title":"Proc. 2nd Conf. Robot Learn.","volume":"87","author":"Vinitsky"},{"key":"ref58","first-page":"267","article-title":"Approximately optimal approximate reinforcement learning","volume-title":"Proc. 19th Int. Conf. Mach. Learn.","author":"Kakade"}],"container-title":["IEEE Transactions on Wireless Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7693\/10384459\/10147312.pdf?arnumber=10147312","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,12]],"date-time":"2024-01-12T02:30:11Z","timestamp":1705026611000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10147312\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1]]},"references-count":58,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/twc.2023.3279268","relation":{},"ISSN":["1536-1276","1558-2248"],"issn-type":[{"value":"1536-1276","type":"print"},{"value":"1558-2248","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1]]}}}