{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T16:44:35Z","timestamp":1759941875951,"version":"3.37.3"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T00:00:00Z","timestamp":1674777600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T00:00:00Z","timestamp":1674777600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s10458-023-09599-5","type":"journal-article","created":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T15:59:08Z","timestamp":1674835148000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Online Markov decision processes with non-oblivious strategic adversary"],"prefix":"10.1007","volume":"37","author":[{"given":"Le Cong","family":"Dinh","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David Henry","family":"Mguni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Tran-Thanh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8132-5613","authenticated-orcid":false,"given":"Yaodong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,1,27]]},"reference":[{"key":"9599_CR1","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R. S., & Barto, A. G. (2018). Reinforcement Learning: An Introduction. Massachusetts: MIT press."},{"issue":"1","key":"9599_CR2","doi-asserted-by":"publisher","first-page":"55","DOI":"10.3233\/KES-2010-0206","volume":"15","author":"GJ Laurent","year":"2011","unstructured":"Laurent, G. J., Matignon, L., Fort-Piat, L., et al. (2011). The world of independent learners is not markovian. International Journal of Knowledge-based and Intelligent Engineering Systems, 15(1), 55\u201364.","journal-title":"International Journal of Knowledge-based and Intelligent Engineering Systems"},{"issue":"3","key":"9599_CR3","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1287\/moor.1090.0396","volume":"34","author":"E Even-Dar","year":"2009","unstructured":"Even-Dar, E., Kakade, S. M., & Mansour, Y. (2009). Online markov decision processes. Mathematics of Operations Research, 34(3), 726\u2013736.","journal-title":"Mathematics of Operations Research"},{"key":"9599_CR4","unstructured":"Dick, T., Gyorgy, A., & Szepesvari, C. (2014). Online learning in markov decision processes with changing cost sequences. In ICML (pp. 512\u2013520)."},{"key":"9599_CR5","unstructured":"Neu, G., Antos, A., Gy\u00f6rgy, A., & Szepesv\u00e1ri, C. (2010). Online markov decision processes under bandit feedback. In NeurIPS (pp. 1804\u20131812)."},{"key":"9599_CR6","unstructured":"Neu, G., & Olkhovskaya, J. (2020). Online learning in mdps with linear function approximation and bandit feedback. arXiv e-prints, 2007."},{"key":"9599_CR7","unstructured":"Yang, Y., & Wang, J. (2020). An overview of multi-agent reinforcement learning from game theoretical perspective. arXiv preprint arXiv:2011.00583"},{"issue":"1\u20132","key":"9599_CR8","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1006\/game.1999.0738","volume":"29","author":"Y Freund","year":"1999","unstructured":"Freund, Y., & Schapire, R. E. (1999). Adaptive game playing using multiplicative weights. Games and Economic Behavior, 29(1\u20132), 79\u2013103.","journal-title":"Games and Economic Behavior"},{"issue":"2","key":"9599_CR9","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1561\/2200000018","volume":"4","author":"S Shalev-Shwartz","year":"2011","unstructured":"Shalev-Shwartz, S., et al. (2011). Online learning and online convex optimization. Foundations and Trends in Machine Learning, 4(2), 107\u2013194.","journal-title":"Foundations and Trends in Machine Learning"},{"key":"9599_CR10","doi-asserted-by":"crossref","unstructured":"Mertikopoulos, P., Papadimitriou, C., & Piliouras, G. (2018). Cycles in adversarial regularized learning. In Proceedings of the twenty-ninth annual ACM-SIAM symposium on discrete algorithms (pp. 2703\u20132717). SIAM.","DOI":"10.1137\/1.9781611975031.172"},{"key":"9599_CR11","doi-asserted-by":"crossref","unstructured":"Bailey, J. P., & Piliouras, G. (2018). Multiplicative weights update in zero-sum games. In Proceedings of the 2018 ACM conference on economics and computation (pp. 321\u2013338).","DOI":"10.1145\/3219166.3219235"},{"key":"9599_CR12","unstructured":"Dinh, L. C., Nguyen, T.-D., Zemhoho, A. B., & Tran-Thanh, L. (2021). Last round convergence and no-dynamic regret in asymmetric repeated games. In Algorithmic learning theory (pp. 553\u2013577) PMLR."},{"key":"9599_CR13","unstructured":"Mertikopoulos, P., Lecouat, B., Zenati, H., Foo, C.-S., Chandrasekhar, V., & Piliouras, G. (2019). Optimistic mirror descent in saddle-point problems: Going the extra (gradient) mile. In ICLR 2019-7th international conference on learning representations (pp. 1\u201323)."},{"key":"9599_CR14","doi-asserted-by":"publisher","first-page":"105095","DOI":"10.1016\/j.jet.2020.105095","volume":"189","author":"DS Leslie","year":"2020","unstructured":"Leslie, D. S., Perkins, S., & Xu, Z. (2020). Best-response dynamics in zero-sum stochastic games. Journal of Economic Theory, 189, 105095.","journal-title":"Journal of Economic Theory"},{"key":"9599_CR15","doi-asserted-by":"crossref","unstructured":"Guan, P., Raginsky, M., Willett, R., & Zois, D.-S. (2016). Regret minimization algorithms for single-controller zero-sum stochastic games. In 2016 IEEE 55th conference on decision and control (CDC) (pp 7075\u20137080). IEEE","DOI":"10.1109\/CDC.2016.7799359"},{"issue":"3","key":"9599_CR16","doi-asserted-by":"publisher","first-page":"676","DOI":"10.1109\/TAC.2013.2292137","volume":"59","author":"G Neu","year":"2013","unstructured":"Neu, G., Gy\u00f6rgy, A., Szepesv\u00e1ri, C., & Antos, A. (2013). Online markov decision processes under bandit feedback. IEEE Transactions on Automatic Control, 59(3), 676\u2013691.","journal-title":"IEEE Transactions on Automatic Control"},{"key":"9599_CR17","doi-asserted-by":"crossref","unstructured":"Filar, J., & Vrieze, K. (1997). Applications and special classes of stochastic games. In Competitive markov decision processes (pp 301\u2013341). Springer, New York.","DOI":"10.1007\/978-1-4612-4054-9_6"},{"key":"9599_CR18","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1016\/S0927-0507(05)80172-0","volume":"2","author":"ML Puterman","year":"1990","unstructured":"Puterman, M. L. (1990). Markov decision processes. Handbooks in Operations Research and Management Science, 2, 331\u2013434.","journal-title":"Handbooks in Operations Research and Management Science"},{"key":"9599_CR19","unstructured":"McMahan, H. B., Gordon, G. J., & Blum, A. (2003). Planning in the presence of cost functions controlled by an adversary. In Proceedings of the 20th international conference on machine learning (ICML-03) (pp. 536\u2013543)."},{"key":"9599_CR20","unstructured":"Dinh, L. C., Yang, Y., Tian, Z., Nieves, N. P., Slumbers, O., Mguni, D. H., & Wang, J. (2021). Online double oracle. arXiv preprint arXiv:2103.07780"},{"key":"9599_CR21","unstructured":"Wei, C.-Y., Hong, Y.-T., & Lu, C.-J. (2017). Online reinforcement learning in stochastic games. arXiv preprint arXiv:1712.00579"},{"key":"9599_CR22","doi-asserted-by":"crossref","unstructured":"Cheung, W. C., Simchi-Levi, D., & Zhu, R. (2019). Non-stationary reinforcement learning: The blessing of (more) optimism. Available at SSRN 3397818.","DOI":"10.2139\/ssrn.3397818"},{"issue":"3","key":"9599_CR23","doi-asserted-by":"publisher","first-page":"737","DOI":"10.1287\/moor.1090.0397","volume":"34","author":"JY Yu","year":"2009","unstructured":"Yu, J. Y., Mannor, S., & Shimkin, N. (2009). Markov decision processes with arbitrary reward processes. Mathematics of Operations Research, 34(3), 737\u2013757.","journal-title":"Mathematics of Operations Research"},{"key":"9599_CR24","unstructured":"Arora, R., Dekel, O., & Tewari, A. (2012). Online bandit learning against an adaptive adversary: from regret to policy regret. arXiv preprint arXiv:1206.6400"},{"key":"9599_CR25","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511546921","volume-title":"Prediction, Learning, and Games","author":"N Cesa-Bianchi","year":"2006","unstructured":"Cesa-Bianchi, N., & Lugosi, G. (2006). Prediction, Learning, and Games. Cambridge: Cambridge University Press."},{"key":"9599_CR26","first-page":"1729","volume":"20","author":"M Zinkevich","year":"2007","unstructured":"Zinkevich, M., Johanson, M., Bowling, M., & Piccione, C. (2007). Regret minimization in games with incomplete information. Advances in Neural Information Processing Systems, 20, 1729\u20131736.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9599_CR27","unstructured":"Daskalakis, C., Ilyas, A., Syrgkanis, V., & Zeng, H. (2017). Training gans with optimism. arXiv preprint arXiv:1711.00141"},{"issue":"10","key":"9599_CR28","doi-asserted-by":"publisher","first-page":"1095","DOI":"10.1073\/pnas.39.10.1095","volume":"39","author":"LS Shapley","year":"1953","unstructured":"Shapley, L. S. (1953). Stochastic games. Proceedings of the National Academy of Sciences, 39(10), 1095\u20131100.","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"9599_CR29","doi-asserted-by":"crossref","unstructured":"Deng, X., Li, Y., Mguni, D. H., Wang, J., & Yang, Y. (2021). On the complexity of computing markov perfect equilibrium in general-sum stochastic games. arXiv preprint arXiv:2109.01795","DOI":"10.1093\/nsr\/nwac256"},{"key":"9599_CR30","unstructured":"Tian, Y., Wang, Y., Yu, T., & Sra, S. (2020). Online learning in unknown markov games."},{"issue":"1","key":"9599_CR31","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1007\/BF01448847","volume":"100","author":"Jv Neumann","year":"1928","unstructured":"Neumann, Jv. (1928). Zur theorie der gesellschaftsspiele. Mathematische Annalen, 100(1), 295\u2013320.","journal-title":"Mathematische Annalen"},{"issue":"1","key":"9599_CR32","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1073\/pnas.36.1.48","volume":"36","author":"JF Nash","year":"1950","unstructured":"Nash, J. F., et al. (1950). Equilibrium points in n-person games. Proceedings of the National Academy of Sciences, 36(1), 48\u201349.","journal-title":"Proceedings of the National Academy of Sciences"},{"issue":"1","key":"9599_CR33","first-page":"374","volume":"13","author":"GW Brown","year":"1951","unstructured":"Brown, G. W. (1951). Iterative solution of games by fictitious play. Activity Analysis of Production and Allocation, 13(1), 374\u2013376.","journal-title":"Activity Analysis of Production and Allocation"},{"key":"9599_CR34","unstructured":"Czarnecki, W. M., Gidel, G., Tracey, B., Tuyls, K., Omidshafiei, S., Balduzzi, D., & Jaderberg, M. (2020). Real world games look like spinning tops. arXiv preprint arXiv:2004.09468"},{"key":"9599_CR35","unstructured":"Perez-Nieves, N., Yang, Y., Slumbers, O., Mguni, D. H., Wen, Y., & Wang, J. (2021). Modelling behavioural diversity for learning in open-ended games. In International conference on machine learning (pp. 8514\u20138524). PMLR"},{"key":"9599_CR36","unstructured":"Liu, X., Jia, H., Wen, Y., Yang, Y., Hu, Y., Chen, Y., Fan, C., & Hu, Z. (2021). Unifying behavioral and response diversity for open-ended learning in zero-sum games. arXiv preprint arXiv:2106.04958"},{"key":"9599_CR37","unstructured":"Yang, Y., Luo, J., Wen, Y., Slumbers, O., Graves, D., Bou\u00a0Ammar, H., Wang, J., & Taylor, M. E. (2021). Diverse auto-curriculum is critical for successful real-world multiagent learning systems. In Proceedings of the 20th international conference on autonomous agents and multiagent systems (pp. 51\u201356)."},{"key":"9599_CR38","first-page":"51","volume":"1","author":"H Bohnenblust","year":"1950","unstructured":"Bohnenblust, H., Karlin, S., & Shapley, L. (1950). Solutions of discrete, two-person games. Contributions to the Theory of Games, 1, 51\u201372.","journal-title":"Contributions to the Theory of Games"},{"issue":"7782","key":"9599_CR39","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., Babuschkin, I., Czarnecki, W. M., Mathieu, M., Dudzik, A., Chung, J., Choi, D. H., Powell, R., Ewalds, T., Georgiev, P., et al. (2019). Grandmaster level in starcraft ii using multi-agent reinforcement learning. Nature, 575(7782), 350\u2013354.","journal-title":"Nature"},{"key":"9599_CR40","unstructured":"Daskalakis, C., & Panageas, I. (2019). Last-iterate convergence: Zero-sum games and constrained min-max optimization. In 10th innovations in theoretical computer science."},{"issue":"1\u20132","key":"9599_CR41","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1007\/s10994-006-0143-1","volume":"67","author":"V Conitzer","year":"2007","unstructured":"Conitzer, V., & Sandholm, T. (2007). Awesome: A general multiagent learning algorithm that converges in self-play and learns a best response against stationary opponents. Machine Learning, 67(1\u20132), 23\u201343.","journal-title":"Machine Learning"},{"issue":"2","key":"9599_CR42","doi-asserted-by":"publisher","first-page":"182","DOI":"10.1007\/s10458-013-9222-4","volume":"28","author":"D Chakraborty","year":"2014","unstructured":"Chakraborty, D., & Stone, P. (2014). Multiagent learning in the presence of memory-bounded agents. Autonomous Agents and Multi-agent Systems, 28(2), 182\u2013213.","journal-title":"Autonomous Agents and Multi-agent Systems"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09599-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-023-09599-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09599-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,11]],"date-time":"2023-05-11T07:42:12Z","timestamp":1683790932000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-023-09599-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,27]]},"references-count":42,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["9599"],"URL":"https:\/\/doi.org\/10.1007\/s10458-023-09599-5","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"type":"print","value":"1387-2532"},{"type":"electronic","value":"1573-7454"}],"subject":[],"published":{"date-parts":[[2023,1,27]]},"assertion":[{"value":"4 January 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 January 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"15"}}