{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T02:31:07Z","timestamp":1730341867582,"version":"3.28.0"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["2144634,2231350"],"award-info":[{"award-number":["2144634,2231350"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,7,10]]},"DOI":"10.23919\/acc60939.2024.10644167","type":"proceedings-article","created":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T17:56:19Z","timestamp":1725558979000},"page":"4032-4037","source":"Crossref","is-referenced-by-count":1,"title":["Oracle Complexity Reduction for Model-Free LQR: A Stochastic Variance-Reduced Policy Gradient Approach"],"prefix":"10.23919","author":[{"given":"Leonardo F.","family":"Toso","sequence":"first","affiliation":[{"name":"Columbia University in the City of New York,Department of Electrical Engineering,New York,NY,USA,10027"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"Wang","sequence":"additional","affiliation":[{"name":"Columbia University in the City of New York,Department of Electrical Engineering,New York,NY,USA,10027"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James","family":"Anderson","sequence":"additional","affiliation":[{"name":"Columbia University in the City of New York,Department of Electrical Engineering,New York,NY,USA,10027"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9992612"},{"key":"ref2","first-page":"1467","article-title":"Global convergence of policy gradient methods for the linear quadratic regulator","volume-title":"International conference on machine learning","author":"Fazel","year":"2018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2020.3037046"},{"key":"ref4","first-page":"2916","article-title":"Derivative-free methods for policy optimization: Guarantees for linear quadratic systems","volume-title":"The 22nd international conference on artificial intelligence and statistics","author":"Malik","year":"2019"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2020.3006256"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-042920-020021"},{"key":"ref7","first-page":"29274","article-title":"Stabilizing dynamical systems via policy gradient methods","volume":"34","author":"Perdomo","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1002\/0471722138"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2008.02.003"},{"key":"ref10","article-title":"Model-free Learning with Heterogeneous Dynamical Systems: A Federated LQR Approach","author":"Wang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2022.110741"},{"key":"ref12","article-title":"Accelerating stochastic gradient descent using predictive variance reduction","volume":"26","author":"Johnson","year":"2013","journal-title":"Advances in neural information processing systems"},{"key":"ref13","first-page":"4026","article-title":"Stochastic variance-reduced policy gradient","volume-title":"International conference on machine learning","author":"Papini","year":"2018"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9029985"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.23919\/ACC45564.2020.9147749"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3087455"},{"key":"ref17","article-title":"A stochastic gradient method with an exponential convergence _rate for finite training sets","volume":"25","author":"Roux","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.5555\/2968826.2969010"},{"key":"ref19","first-page":"541","article-title":"An improved convergence analysis of stochastic variance-reduced policy gradient","volume-title":"Uncertainty in Artificial Intelligence","author":"Xu","year":"2020"},{"key":"ref20","first-page":"7624","article-title":"An improved analysis of (variance-reduced) policy gradient and natural policy gradient methods","volume":"33","author":"Liu","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref21","first-page":"3100","article-title":"Improved zeroth-order variance reduced algorithms and analysis for nonconvex optimization","volume-title":"International conference on machine learning","author":"Ji","year":"2019"},{"key":"ref22","article-title":"Zeroth-order stochastic variance reduction for nonconvex optimization","volume":"31","author":"Liu","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.1971.1099755"},{"key":"ref24","first-page":"531","article-title":"Learning the model-free linear quadratic regulator via random search","volume-title":"Learning for Dynamics and Control","author":"Mohammadi","year":"2020"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.23919\/acc60939.2024.10644167"},{"key":"ref26","first-page":"6050","article-title":"Stem: A stochastic two-sided momentum algorithm achieving near-optimal sample and communication complexities for federated learning","volume":"34","author":"Khanduri","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref27","article-title":"Spider: Near-optimal non-convex optimization via stochastic path-integrated differential estima-tor","volume":"31","author":"Fang","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/s10208-011-9099-z"}],"event":{"name":"2024 American Control Conference (ACC)","start":{"date-parts":[[2024,7,10]]},"location":"Toronto, ON, Canada","end":{"date-parts":[[2024,7,12]]}},"container-title":["2024 American Control Conference (ACC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10644130\/10644150\/10644167.pdf?arnumber=10644167","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,7]],"date-time":"2024-09-07T06:43:32Z","timestamp":1725691412000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10644167\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,10]]},"references-count":28,"URL":"https:\/\/doi.org\/10.23919\/acc60939.2024.10644167","relation":{},"subject":[],"published":{"date-parts":[[2024,7,10]]}}}