{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T17:52:25Z","timestamp":1787507545641,"version":"build-2736575974"},"reference-count":25,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1109\/cdc.2016.7798956","type":"proceedings-article","created":{"date-parts":[[2017,1,5]],"date-time":"2017-01-05T12:11:18Z","timestamp":1483618278000},"page":"4516-4521","source":"Crossref","is-referenced-by-count":7,"title":["An online primal-dual method for discounted Markov decision processes"],"prefix":"10.1109","author":[{"given":"Mengdi","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yichen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","first-page":"32611","article-title":"Randomized first-order methods for saddle point optimization","author":"dang","year":"2014","journal-title":"Manuscript"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1287\/opre.51.6.850.24925"},{"key":"ref12","author":"kushner","year":"2003","journal-title":"Stochastic Approximation and Recursive Algorithms and Applications"},{"key":"ref13","first-page":"836","article-title":"Regularized off-policy td-learning","author":"liu","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s10109-012-0166-z"},{"key":"ref15","author":"mahadevan","year":"2012","journal-title":"Sparse q-learning with mirror descent"},{"key":"ref16","author":"mahadevan","year":"2014","journal-title":"Proximal reinforcement learning A new theory of sequential decision making in primal-dual spaces"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/0-306-48102-2_8"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1287\/educ.2013.0118"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/1921598.1921603"},{"key":"ref4","volume":"1","author":"bertsekas","year":"1995","journal-title":"Dynamic Programming and Optimal Control"},{"key":"ref3","author":"benveniste","year":"2012","journal-title":"Adaptive Algorithms and Stochastic Approximations"},{"key":"ref6","author":"bertsekas","year":"1989","journal-title":"Parallel and Distributed Computation Numerical Methods"},{"key":"ref5","first-page":"560","article-title":"Neuro-dynamic programming: an overview. In Decision and Control, 1995","volume":"1","author":"bertsekas","year":"1995","journal-title":"Proceedings of the 34th IEEE Conference on"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1137\/130919362"},{"key":"ref7","doi-asserted-by":"crossref","DOI":"10.1007\/978-93-86279-38-5","author":"borkar","year":"2008","journal-title":"Stochastic Approximation A Dynamical Systems Viewpoint"},{"key":"ref2","author":"azar","year":"2012","journal-title":"On the theory of reinforcement learning methods convergence analysis and sample complexity"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ECC.2015.7330554"},{"key":"ref1","author":"abbasi-yadkori","year":"2014","journal-title":"Linear programming for large-scale markov decision problems"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1137\/0330046"},{"key":"ref22","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref21","author":"puterman","year":"2014","journal-title":"Markov Decision Processes Discrete Stochastic Dynamic Programming"},{"key":"ref24","first-page":"631","article-title":"Dual temporal difference learning","author":"yang","year":"2009","journal-title":"International Conference on Artificial Intelligence and Statistics"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1120.0574"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1287\/moor.1110.0516"}],"event":{"name":"2016 IEEE 55th Conference on Decision and Control (CDC)","location":"Las Vegas, NV, USA","start":{"date-parts":[[2016,12,12]]},"end":{"date-parts":[[2016,12,14]]}},"container-title":["2016 IEEE 55th Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7786694\/7798233\/07798956.pdf?arnumber=7798956","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,17]],"date-time":"2019-09-17T01:39:23Z","timestamp":1568684363000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7798956\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,12]]},"references-count":25,"URL":"https:\/\/doi.org\/10.1109\/cdc.2016.7798956","relation":{},"subject":[],"published":{"date-parts":[[2016,12]]}}}