{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T18:57:31Z","timestamp":1767207451703,"version":"3.48.0"},"reference-count":37,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1109\/tac.2025.3589416","type":"journal-article","created":{"date-parts":[[2025,7,15]],"date-time":"2025-07-15T17:45:19Z","timestamp":1752601519000},"page":"275-290","source":"Crossref","is-referenced-by-count":1,"title":["Markov $\\alpha$-Potential Games"],"prefix":"10.1109","volume":"71","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3350-4606","authenticated-orcid":false,"given":"Xin","family":"Guo","sequence":"first","affiliation":[{"name":"Department of Industrial Engineering and Operations Research, University of California, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1206-716X","authenticated-orcid":false,"given":"Xinyu","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Industrial Engineering and Operations Research, University of California, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3596-2851","authenticated-orcid":false,"given":"Chinmay","family":"Maheshwari","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, Johns Hopkins University, Baltimore, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1300-1574","authenticated-orcid":false,"given":"Shankar","family":"Sastry","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering and Computer Sciences, University of California, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5334-4163","authenticated-orcid":false,"given":"Manxi","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Civil and Environmental Engineering, University of California, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.39.10.1953"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60990-0_12"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511800481.004"},{"key":"ref4","article-title":"Decentralized Q-learning in zero-sum Markov games","volume-title":"Proc. 35th Int. Conf. Neural Inf. Process. Syst.","author":"Sayin","year":"2021"},{"key":"ref5","first-page":"1664","article-title":"Fictitious play and best-response dynamics in identical interest and zero-SUM stochastic games","volume-title":"Int. Conf. Mach. Learn.","author":"Baudin","year":"2022"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1137\/22M1515112"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/tac.2025.3576379"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2022.11.031"},{"key":"ref9","first-page":"4414","article-title":"Global convergence of multi-agent policy gradient in Markov potential games","volume-title":"Proc. ICLR Workshop Gamification Multiagent Solutions","author":"Leonardos","year":"2022"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9992762"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2016.2598476"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3121228"},{"key":"ref13","article-title":"Learning parametric closed-loop policies for Markov potential games","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Macua","year":"2018"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1006\/game.1996.0044"},{"key":"ref15","first-page":"5166","article-title":"Independent policy gradient for large-scale Markov potential games: Sharper rates, function approximation, and game-agnostic convergence","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ding","year":"2022"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/2465769.2465776"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1110.0500"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/BF01737559"},{"key":"ref19","first-page":"11009","article-title":"Learning in congestion games with bandit feedback","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst.","author":"Cui","year":"2022"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.23919\/ACC60939.2024.10644523"},{"article-title":"Imagined potential games: A framework for simulating, learning and evaluating interactive behaviors","year":"2024","author":"Sun","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2021.XVII.084"},{"article-title":"Decentralized cooperative multi-agent reinforcement learning with exploration","year":"2021","author":"Mao","key":"ref23"},{"key":"ref24","article-title":"When can we learn general-SUM Markov games with a large number of players sample-efficiently?","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Song","year":"2021"},{"key":"ref25","first-page":"4414","article-title":"Independent natural policy gradient always converges in Markov potential games","volume-title":"Proc. 25th Int. Conf. Artif. Intell. Statist.","author":"Fox","year":"2022"},{"key":"ref26","first-page":"43951","article-title":"Provably fast convergence of independent natural policy gradient for Markov potential games","volume-title":"Proc. NeurIPS","author":"Sun","year":"2023"},{"key":"ref27","first-page":"1923","article-title":"On the global convergence rates of decentralized softmax gradient play in Markov potential games","volume-title":"Proc. 36th Int. Conf. Neural Inf. Process. Syst.","author":"Zhang","year":"2022"},{"article-title":"Algorithms can learn to collude: A folk theorem from learning with bounded rationality","year":"2022","author":"Cartea","key":"ref28"},{"key":"ref29","first-page":"27952","article-title":"Fast policy extragradient methods for competitive games with entropy regularization","volume-title":"Proc. 35th Int. Conf. Neural Inf. Process. Syst.","author":"Cen","year":"2021"},{"key":"ref30","first-page":"534","article-title":"Learning and calibrating heterogeneous bounded rational market behaviour with multi-agent reinforcement learning","volume-title":"Proc. 23rd Int. Conf. Auton. Agents Multiagent Syst.","author":"Evans","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2016.0778"},{"key":"ref32","first-page":"5527","article-title":"Independent policy gradient methods for competitive reinforcement learning","volume-title":"Proc. 34th Conf. Neural Inf. Process. Syst.","author":"Daskalakis","year":"2020"},{"key":"ref33","volume-title":"The Theory of Learning in Games","volume":"2","author":"Fudenberg","year":"1998"},{"key":"ref34","volume-title":"Principles of Mathematical Analysis","volume":"3","author":"Rudin","year":"1976"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.23919\/ECC.2003.7086500"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1137\/1035089"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1137\/24m1707316"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9\/11319285\/11080281.pdf?arnumber=11080281","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T18:47:44Z","timestamp":1767206864000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11080281\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":37,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tac.2025.3589416","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"type":"print","value":"0018-9286"},{"type":"electronic","value":"1558-2523"},{"type":"electronic","value":"2334-3303"}],"subject":[],"published":{"date-parts":[[2026,1]]}}}