{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T00:37:59Z","timestamp":1770683879666,"version":"3.49.0"},"reference-count":60,"publisher":"Society for Industrial & Applied Mathematics (SIAM)","issue":"1","funder":[{"DOI":"10.13039\/501100023650","name":"National Center of Competence in Research","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100023650","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001711","name":"Schweizerischer Nationalfonds zur F\u00f6rderung der Wissenschaftlichen Forschung","doi-asserted-by":"publisher","award":["51NF40_180545"],"award-info":[{"award-number":["51NF40_180545"]}],"id":[{"id":"10.13039\/501100001711","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["SIAM J. Optim."],"published-print":{"date-parts":[[2026,3,31]]},"DOI":"10.1137\/24m1631250","type":"journal-article","created":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T08:21:16Z","timestamp":1770625276000},"page":"120-151","source":"Crossref","is-referenced-by-count":0,"title":["Policy Gradient Algorithms for Robust MDP\\(\\text{s}\\) with Nonrectangular Uncertainty Sets"],"prefix":"10.1137","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2978-7949","authenticated-orcid":true,"given":"Mengmeng","family":"Li","sequence":"first","affiliation":[{"name":"Risk Analytics and Optimization Chair, EPFL, 1015 Lausanne, Switzerland."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2697-8886","authenticated-orcid":true,"given":"Daniel","family":"Kuhn","sequence":"additional","affiliation":[{"name":"Risk Analytics and Optimization Chair, EPFL, 1015 Lausanne, Switzerland."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tobias","family":"Sutter","sequence":"additional","affiliation":[{"name":"Department of Economics, University of St. Gallen, 9000 St. Gallen, Switzerland."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"351","published-online":{"date-parts":[[2026,2,9]]},"reference":[{"key":"ref1","first-page":"1","volume":"22","author":"Agarwal A.","year":"2021","journal-title":"J. Mach. Learn. Res."},{"key":"ref2","unstructured":"J. Altschuler and K. Talwar, Concentration of the Langevin Algorithm\u2019s Stationary DIstribution, preprint, arXiv:2212.12629, 2022."},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1057\/jors.1995.50"},{"key":"ref4","volume-title":"Topological Spaces: Including a Treatment of Multi-Valued Functions, Vector Spaces, and Convexity","author":"Berge C.","year":"1997"},{"key":"ref5","volume-title":"Nonlinear Programming","author":"Bertsekas D.","year":"2016"},{"key":"ref6","volume-title":"Neuro-Dynamic Programming","author":"Bertsekas D. P.","year":"1996"},{"key":"ref7","unstructured":"J. Bhandari and D. Russo, On the linear convergence of policy gradient methods for finite MDPs, in Proceedings of the International Conference on Artificial Intelligence and Statistics, 2021."},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2009.07.008"},{"key":"ref9","volume-title":"Statistical Inference for Markov Processes","author":"Billingsley P.","year":"1961"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s00780-014-0234-y"},{"key":"ref11","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Blanchet J.","year":"2023"},{"key":"ref12","unstructured":"J. Chae, S. Han, W. Jung, M. Cho, S. Choi, and Y. Sung, Robust imitation learning against variations in environment dynamics, in Proceedings of the International Conference on Machine Learning, 2022."},{"key":"ref13","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Daskalakis C.","year":"2020"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1137\/18M1178244"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1287\/opre.1080.0685"},{"key":"ref16","doi-asserted-by":"crossref","unstructured":"J. Duchi, S. Shalev-Shwartz, Y. Singer, and T. Chandra, Efficient projections onto the \\(\\ell_1\\)-ball for learning in high dimensions, in Proceedings of the International Conference on Machine Learning, 2008.","DOI":"10.1145\/1390156.1390191"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1002\/nav.3800030109"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2017.1685"},{"key":"ref19","volume-title":"Proceedings of the 2nd Conference on Learning for Dynamics and Control","author":"Gong H.","year":"2020"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2022.1259"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2022.0284"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-0729-0"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1214\/aop\/1176994579"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1040.0129"},{"key":"ref25","unstructured":"S. Kakade and J. Langford, Approximately optimal approximate reinforcement learning, in Proceedings of the International Conference on Machine Learning, 2002."},{"key":"ref26","unstructured":"A. Lamperski, Projected stochastic gradient Langevin algorithms for constrained sampling and non-convex learning, in Proceedings of the Conference on Learning Theory, 2021."},{"key":"ref27","unstructured":"Y. Le Tallec, Robust, Risk-Sensitive, and Data-Driven Control of Markov Decision Processes, Ph.D. thesis, Massachusetts Institute of Technology, 2007."},{"key":"ref28","volume":"11","author":"Lesmana N.","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref29","unstructured":"M. Li, T. Sutter, and D. Kuhn, Distributionally robust optimization with Markovian data, in Proceedings of the International Conference on Machine Learning, 2021."},{"key":"ref30","unstructured":"Y. Li and G. Lan, First-Order Policy Optimization for Robust Policy Evaluation, preprint, arXiv:2307.15890, 2023."},{"key":"ref31","unstructured":"Y. Li, T. Zhao, and G. Lan, First-Order Policy Optimization for Robust Markov Decision Process, preprint, arXiv:2209.10579, 2022."},{"key":"ref32","volume-title":"Statistical Decision Theory: Estimation, Testing, and Selection","author":"Liese F.","year":"2008"},{"key":"ref33","first-page":"1","volume-title":"J. Mach. Learn. Res.","volume":"26","author":"Lin T.","year":"2025"},{"key":"ref34","doi-asserted-by":"crossref","unstructured":"J. Liu and J. Ye, Efficient Euclidean projections in linear time, in Proceedings of the International Conference on Machine Learning, 2009.","DOI":"10.1145\/1553374.1553459"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.2307\/2297651"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2016.0786"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1017\/9781009051873"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1287\/opre.1050.0216"},{"key":"ref39","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"Puterman M.","year":"2005"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.2307\/3318418"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1515\/9781400873173"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.orl.2012.08.007"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2015.1466"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3582560"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2021.0609"},{"key":"ref46","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton R.","year":"2018"},{"key":"ref47","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Sutton R.","year":"1999"},{"key":"ref48","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Thekumparampil K. K.","year":"2019"},{"key":"ref49","unstructured":"I. Usmanova, M. Kamgarpour, A. Krause, and K. Levy, Fast projection onto convex smooth constraints, in Proceedings of the International Conference on Machine Learning, 2021."},{"key":"ref50","doi-asserted-by":"crossref","unstructured":"L. Viano, Y.T. Huang, P. Kamalaruban, C. Innes, S. Ramamoorthy, and A. Weller, Robust learning from observation with model misspecification, in Proceedings of the International Conference on Autonomous Agents and Multiagent Systems, 2022.","DOI":"10.65109\/MGKH7405"},{"key":"ref51","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"Viano L.","year":"2021"},{"key":"ref52","unstructured":"J. Wang, J. Zhang, H. Jiang, J. Zhang, L. Wang, and C. Zhang, Offline meta reinforcement learning with in-distribution online adaptation, in Proceedings of the International Conference on Machine Learning, 2023."},{"key":"ref53","unstructured":"Q. Wang, C. P. Ho, and M. Petrik, Policy gradient in robust MDPs with global convergence guarantee, in Procedings of the International Conference on Machine Learning, 2023."},{"key":"ref54","unstructured":"Q. Wang, S. Xu, C. P. Ho, and M. Petrik, Policy Gradient for Robust Markov Decision Processes, preprint, arXiv:2410.22114, 2024."},{"key":"ref55","unstructured":"W. Wang and M. A. Carreira-Perpin\u00e1n, Projection onto the Probability Simplex: An Efficient Algorithm with a Simple Proof, and an Application, preprint, https:\/\/arxiv.org\/abs\/1309.1541, 2013."},{"key":"ref56","unstructured":"Y. Wang and S. Zou, Policy gradient method for robust reinforcement learning, in Proceedings of the International Conference on Machine Learning, 2022."},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1287\/opre.42.4.739"},{"key":"ref58","author":"Wiesemann W.","year":"2023","journal-title":"private communication"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1120.0566"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1137\/22M1477659"}],"container-title":["SIAM Journal on Optimization"],"original-title":[],"language":"en","deposited":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T08:21:25Z","timestamp":1770625285000},"score":1,"resource":{"primary":{"URL":"https:\/\/epubs.siam.org\/doi\/10.1137\/24M1631250"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,9]]},"references-count":60,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,3,31]]}},"alternative-id":["10.1137\/24M1631250"],"URL":"https:\/\/doi.org\/10.1137\/24m1631250","relation":{},"ISSN":["1052-6234","1095-7189"],"issn-type":[{"value":"1052-6234","type":"print"},{"value":"1095-7189","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,9]]}}}