{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T13:06:56Z","timestamp":1769778416928,"version":"3.49.0"},"reference-count":51,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Collaborative Research: Transferable"},{"name":"Hierarchical"},{"name":"Expressive"},{"DOI":"10.13039\/100026216","name":"Optimal","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100026216","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Robust"},{"name":"Interpretable NETworks","award":["DMS-2031899"],"award-info":[{"award-number":["DMS-2031899"]}]},{"name":"Digital Transformation Institute"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1109\/tac.2025.3576379","type":"journal-article","created":{"date-parts":[[2025,6,3]],"date-time":"2025-06-03T13:55:07Z","timestamp":1748958907000},"page":"7538-7553","source":"Crossref","is-referenced-by-count":2,"title":["Independent and Decentralized Learning in Markov Potential Games"],"prefix":"10.1109","volume":"70","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3596-2851","authenticated-orcid":false,"given":"Chinmay","family":"Maheshwari","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering and Computer Sciences, University of California Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5334-4163","authenticated-orcid":false,"given":"Manxi","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Civil and Environmental Engineering, University of California Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0313-1551","authenticated-orcid":false,"given":"Druv","family":"Pai","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Computer Sciences, University of California Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shankar","family":"Sastry","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Computer Sciences, University of California Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Safe, multi-agent, reinforcement learning for autonomous driving","author":"Shalev-Shwartz","year":"2016"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC.2014.6958095"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-008-9062-9"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/S0165-1889(02)00122-7"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1126\/science.aao1733"},{"key":"ref7","article-title":"Global convergence of multi-agent policy gradient in Markov potential games","volume-title":"Proc. Workshop Gamification Multiagent Solutions","author":"Leonardos","year":"2022"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2024.3387208"},{"key":"ref9","article-title":"When can we learn general-sum Markov games with a large number of players sample-efficiently?","author":"Song","year":"2021"},{"key":"ref10","article-title":"Decentralized cooperative multi-agent reinforcement learning with exploration","author":"Mao","year":"2021"},{"key":"ref11","first-page":"5166","article-title":"Independent policy gradient for large-scale Markov potential games: Sharper rates, function approximation, and game-agnostic convergence","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ding","year":"2022"},{"key":"ref12","first-page":"4414","article-title":"Independent natural policy gradient always converges in Markov potential games","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Fox","year":"2022"},{"key":"ref13","first-page":"1923","article-title":"On the global convergence rates of decentralized softmax gradient play in Markov potential games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang","year":"2022"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2012.08.037"},{"key":"ref15","article-title":"Learning parametric closed-loop policies for Markov potential games","author":"Macua","year":"2018"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2016.2598476"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9992762"},{"key":"ref18","first-page":"5527","article-title":"Independent policy gradient methods for competitive reinforcement learning","volume":"33","author":"Daskalakis","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2011.31"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.geb.2005.08.005"},{"key":"ref21","first-page":"1008","article-title":"Actor-critic algorithms","volume":"12","author":"Konda","year":"1999","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1017\/S0269888912000057"},{"key":"ref24","first-page":"860","article-title":"Policy-gradient algorithms have no guarantees of convergence in linear quadratic games","volume-title":"Proc. 19th Int. Conf. Auton. Agents MultiAgent Syst.","author":"Mazumdar","year":"2020"},{"key":"ref25","volume-title":"Stochastic Approximation: A Dynamical Systems Viewpoint","volume":"48","author":"Borkar","year":"2009"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993306"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1287\/11-SSY056"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1137\/21M1426675"},{"key":"ref29","first-page":"18320","article-title":"Decentralized Q-learning in zero-sum Markov games","volume":"34","author":"Sayin","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref30","first-page":"307","article-title":"A natural actor-critic framework for zero-sum Markov games","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Alacaoglu","year":"2022"},{"key":"ref31","first-page":"3899","article-title":"Decentralized single-timescale actor-critic on zero-sum two-player stochastic games","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guo","year":"2021"},{"key":"ref32","first-page":"1321","article-title":"Approximate dynamic programming for two-player zero-sum Markov games","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Perolat","year":"2015"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3490486.3538289"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1142\/S0219525902000535"},{"key":"ref35","first-page":"919","article-title":"Actor-critic fictitious play in simultaneous move multistage games","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Perolat","year":"2018"},{"key":"ref36","first-page":"1371","article-title":"Two-timescale algorithms for learning Nash equilibria in general-sum stochastic games","volume-title":"Proc. 2015 Int. Conf. Auton. Agents Multiagent Syst.","author":"Prasad","year":"2015"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1137\/22M1515112"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1006\/game.1996.0044"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1137\/17M1139461"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1006\/jeth.1996.0014"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1111\/j.1468-0262.2002.00440.x"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2008.2010885"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/2940716.2940784"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1090\/S0273-0979-03-00988-1"},{"key":"ref45","first-page":"6369","article-title":"Learning with bandit feedback in potential games","volume":"30","author":"Heliou","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref46","first-page":"163","article-title":"On the convergence of no-regret learning in selfish routing","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Krichene","year":"2014"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1016\/j.geb.2008.11.012"},{"key":"ref48","volume-title":"Game Theory","author":"Fudenberg","year":"1991"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.jet.2020.105095"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/BFb0096509"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1137\/S0363012904439301"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/9\/11218261\/11023106-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9\/11218261\/11023106.pdf?arnumber=11023106","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T21:28:22Z","timestamp":1769722102000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11023106\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":51,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tac.2025.3576379","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"value":"0018-9286","type":"print"},{"value":"1558-2523","type":"electronic"},{"value":"2334-3303","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11]]}}}