{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T14:14:39Z","timestamp":1785420879773,"version":"3.56.0"},"reference-count":50,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2022,12,1]],"date-time":"2022-12-01T00:00:00Z","timestamp":1669852800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,12,1]],"date-time":"2022-12-01T00:00:00Z","timestamp":1669852800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,12,1]],"date-time":"2022-12-01T00:00:00Z","timestamp":1669852800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"NSF CAREER ECCS","award":["1553407"],"award-info":[{"award-number":["1553407"]}]},{"name":"NSF AI institute","award":["2112085"],"award-info":[{"award-number":["2112085"]}]},{"name":"NSF CNS","award":["2003111"],"award-info":[{"award-number":["2003111"]}]},{"name":"AFOSR YIP","award":["FA9550-18-1-0150"],"award-info":[{"award-number":["FA9550-18-1-0150"]}]},{"name":"ONR YIP","award":["N00014-19-1-2217"],"award-info":[{"award-number":["N00014-19-1-2217"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1109\/tac.2021.3128592","type":"journal-article","created":{"date-parts":[[2021,11,16]],"date-time":"2021-11-16T20:28:01Z","timestamp":1637094481000},"page":"6429-6444","source":"Crossref","is-referenced-by-count":64,"title":["Distributed Reinforcement Learning for Decentralized Linear Quadratic Control: A Derivative-Free Policy Optimization Approach"],"prefix":"10.1109","volume":"67","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1858-4257","authenticated-orcid":false,"given":"Yingying","family":"Li","sequence":"first","affiliation":[{"name":"John A. Paulson School of Engineering and Applied Sciences, Harvard University, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4921-8372","authenticated-orcid":false,"given":"Yujie","family":"Tang","sequence":"additional","affiliation":[{"name":"John A. Paulson School of Engineering and Applied Sciences, Harvard University, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9333-3554","authenticated-orcid":false,"given":"Runyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"John A. Paulson School of Engineering and Applied Sciences, Harvard University, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9545-3050","authenticated-orcid":false,"given":"Na","family":"Li","sequence":"additional","affiliation":[{"name":"John A. Paulson School of Engineering and Applied Sciences, Harvard University, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1","article-title":"On the theory of policy gradient methods: Optimality, approximation, and distribution shift","volume":"22","author":"agarwal","year":"2021","journal-title":"J Mach Learn Res"},{"key":"ref38","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume":"32","author":"silver","year":"0","journal-title":"Proc 31st Int Conf Mach Learn"},{"key":"ref33","first-page":"489","article-title":"Learning to cooperate via policy search","author":"peshkin","year":"0","journal-title":"Proc 16th Conf Uncertainty Artif Intell"},{"key":"ref32","first-page":"2681","article-title":"Deep decentralized multi-task multi-agent reinforcement learning under partial observability","volume":"70","author":"omidshafiei","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1287\/moor.27.4.819.297"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2011.6160938"},{"key":"ref37","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume":"12","author":"sutton","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref35","article-title":"Cooperative multi-agent reinforcement learning with partial observations","author":"zhang","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1137\/20M1329858"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1137\/120880811"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2009.5400233"},{"key":"ref2","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of Go without human knowledge","volume":"550","author":"silver","year":"2017","journal-title":"Nature"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-009-9120-4"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2019.8814438"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2014.10.047"},{"key":"ref21","article-title":"LQR through the lens of first order methods: Discrete-time case","author":"bu","year":"2019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2019.8814803"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2016.2542134"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2019.8814952"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619423"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1214\/ECP.v17-2079"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/GlobalSIP.2016.7905980"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1137\/0306011"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1137\/20M1347942"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2005.860365"},{"key":"ref13","first-page":"1","article-title":"Derivative-free methods for policy optimization: Guarantees for linear quadratic systems","volume":"21","author":"malik","year":"2020","journal-title":"J Mach Learn Res"},{"key":"ref14","author":"\u00e5str\u00f6m","year":"2008","journal-title":"Adaptive Control"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1002\/0471669784"},{"key":"ref16","first-page":"8351","article-title":"Provably global convergence of actor-critic: A case for linear quadratic regulator with ergodic cost","volume":"32","author":"yang","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2020.3006256"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3087455"},{"key":"ref19","first-page":"10 154","article-title":"Certainty equivalence is efficient for linear quadratic control","volume":"32","author":"mania","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s10208-019-09426-y"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"621","DOI":"10.1007\/978-3-319-67361-5_40","article-title":"Airsim: High-fidelity visual and physical simulation for autonomous vehicles","volume":"5","author":"shah","year":"2018","journal-title":"Field and Service Robotics Ser Springer Proceedings in Advanced Robotics"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ALLERTON.2017.8262873"},{"key":"ref5","first-page":"1467","article-title":"Global convergence of policy gradient methods for the linear quadratic regulator","volume":"80","author":"fazel","year":"0","journal-title":"Proc 35th Int Conf Mach Learn ser Proc Mach Learn Res"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-008-9062-9"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1002\/9781118122631"},{"key":"ref49","doi-asserted-by":"crossref","DOI":"10.1002\/9780471722199","author":"seber","year":"2003","journal-title":"Linear Regression Analysis"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/PSCE.2009.4840087"},{"key":"ref46","first-page":"314","article-title":"Stochastic variance reduction for nonconvex optimization","volume":"48","author":"reddi","year":"0","journal-title":"Proc 33rd Int Conf Mach Learn"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TCNS.2017.2698261"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/9.59806"},{"key":"ref47","article-title":"Improving the convergence rate of one-point zeroth-order optimization using residual feedback","author":"zhang","year":"2020"},{"key":"ref42","first-page":"385","article-title":"Online convex optimization in the bandit setting: Gradient descent without a gradient","author":"flaxman","year":"0","journal-title":"Proc 16th Annu ACM-SIAM Symp Discrete Algorithms"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.23919\/ACC45564.2020.9147571"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysconle.2004.02.022"},{"key":"ref43","first-page":"3","article-title":"On the complexity of bandit and derivative-free stochastic convex optimization","volume":"30","author":"shamir","year":"0","journal-title":"Proc 26th Annu Conf Learn Theory"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9\/9969927\/09616447.pdf?arnumber=9616447","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T19:33:35Z","timestamp":1672083215000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9616447\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12]]},"references-count":50,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tac.2021.3128592","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"value":"0018-9286","type":"print"},{"value":"1558-2523","type":"electronic"},{"value":"2334-3303","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12]]}}}