{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T22:35:01Z","timestamp":1781735701840,"version":"3.54.5"},"reference-count":54,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,12,6]]},"DOI":"10.1109\/cdc51059.2022.9993175","type":"proceedings-article","created":{"date-parts":[[2023,1,10]],"date-time":"2023-01-10T14:26:56Z","timestamp":1673360816000},"page":"2833-2838","source":"Crossref","is-referenced-by-count":7,"title":["Independent Natural Policy Gradient Methods for Potential Games: Finite-time Global Convergence with Entropy Regularization"],"prefix":"10.1109","author":[{"given":"Shicong","family":"Cen","sequence":"first","affiliation":[{"name":"Carnegie Mellon University,Department of Electrical and Computer Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fan","family":"Chen","sequence":"additional","affiliation":[{"name":"Peking University,Department of Mathematics"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuejie","family":"Chi","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University,Department of Electrical and Computer Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"64","article-title":"Optimality and approximation with policy gradient methods in Markov decision processes","volume-title":"Conference on Learning Theory","author":"Agarwal"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3519935.3520031"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.4086\/toc.2012.v008a006"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2016.2598476"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2021.0014"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9993175"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2021.2151"},{"key":"ref8","article-title":"Fast policy extragradient methods for competitive games with entropy regularization","volume":"34","author":"Cen","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.geb.2009.05.004"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.tcs.2012.02.033"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/2483699.2483703"},{"key":"ref12","article-title":"Near-optimal no-regret learning in general games","volume":"34","author":"Daskalakis","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref13","first-page":"5527","article-title":"Independent policy gradient methods for competitive reinforcement learning","volume":"33","author":"Daskalakis","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref14","article-title":"Independent policy gradient for large-scale markov potential games: Sharper rates, function approximation, and game-agnostic convergence","author":"Ding","year":"2022"},{"key":"ref15","first-page":"1467","article-title":"Global convergence of policy gradient methods for the linear quadratic regulator","volume-title":"International Conference on Machine Learning","author":"Fazel"},{"key":"ref16","article-title":"Independent natural policy gradient always converges in Markov potential games","author":"Fox","year":"2021"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1006\/game.1999.0738"},{"key":"ref18","article-title":"Soft actorcritic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018"},{"key":"ref19","article-title":"Learning with bandit feedback in potential games","volume":"30","author":"Heliou","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1093\/oso\/9780198503682.001.0001"},{"key":"ref21","article-title":"V-learning\u2013a simple, efficient, decentralized algorithm for multiagent RL","author":"Jin","year":"2021"},{"key":"ref22","article-title":"A natural policy gradient","volume":"14","author":"Kakade","year":"2001","journal-title":"Advances in neural information processing systems"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-022-01816-5"},{"key":"ref24","article-title":"Global convergence of multi-agent policy gradient in Markov potential games","author":"Leonardos","year":"2021"},{"issue":"1","key":"ref25","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"Levine","year":"2016","journal-title":"The Journal of Machine Learning Research"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-022-01920-6"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1006\/inco.1994.1009"},{"key":"ref28","article-title":"An improved analysis of (variance-reduced) policy gradient and natural policy gradient methods","volume":"33","author":"Liu","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref29","article-title":"Multi-agent actor-critic for mixed cooperativecompetitive environments","volume":"30","author":"Lowe","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/s13235-021-00420-0"},{"key":"ref31","first-page":"15007","article-title":"On improving model-free algorithms for decentralized multi-agent reinforcement learning","volume-title":"International Conference on Machine Learning","author":"Mao"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/1329125.1329175"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1137\/070680199"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1006\/game.1995.1023"},{"key":"ref35","article-title":"Escaping the gravitational pull of softmax","volume":"33","author":"Mei","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref36","first-page":"6820","article-title":"On the global convergence rates of softmax policy gradient methods","volume-title":"International Conference on Machine Learning","author":"Mei"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2016.0778"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-03544-w"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1006\/jeth.1996.0014"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1006\/game.1996.0044"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.2307\/1969529"},{"key":"ref42","article-title":"Multiplicative weights update with constant step-size in congestion games: Convergence, limit cycles and chaos","volume":"30","author":"Palaiopanos","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1016\/0899-8256(91)90003-w"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref45","article-title":"When can we learn generalsum Markov games with a large number of players sampleefficiently?","author":"Song","year":"2021"},{"key":"ref46","article-title":"Neural policy gradient methods: Global optimality and rates of convergence","author":"Wang","year":"2019"},{"key":"ref47","article-title":"Last-iterate convergence of decentralized optimistic gradient descent\/ascent in infinite-horizon competitive Markov games","author":"Wei","year":"2021"},{"key":"ref48","article-title":"On the convergence rates of policy gradient methods","author":"Xiao","year":"2022"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199269181.001.0001"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctv10h9d35"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1137\/21m1456789"},{"key":"ref52","article-title":"On the effect of log-barrier regularization in decentralized softmax gradient play in multiagent systems","author":"Zhang","year":"2022"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2022.11.031"},{"key":"ref54","article-title":"Provably efficient policy gradient methods for two-player zero-sum Markov games","author":"Zhao","year":"2021"}],"event":{"name":"2022 IEEE 61st Conference on Decision and Control (CDC)","location":"Cancun, Mexico","start":{"date-parts":[[2022,12,6]]},"end":{"date-parts":[[2022,12,9]]}},"container-title":["2022 IEEE 61st Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9992315\/9992317\/09993175.pdf?arnumber=9993175","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T22:06:18Z","timestamp":1710367578000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9993175\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,6]]},"references-count":54,"URL":"https:\/\/doi.org\/10.1109\/cdc51059.2022.9993175","relation":{},"subject":[],"published":{"date-parts":[[2022,12,6]]}}}