{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,28]],"date-time":"2025-10-28T10:43:35Z","timestamp":1761648215861,"version":"3.37.3"},"reference-count":47,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2015,3,1]],"date-time":"2015-03-01T00:00:00Z","timestamp":1425168000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"DOI":"10.13039\/100000181","name":"AFOSR","doi-asserted-by":"crossref","id":[{"id":"10.13039\/100000181","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation (NSF)","doi-asserted-by":"crossref","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"crossref"}]},{"name":"ONR"},{"name":"Center for Dynamic Data Analysis"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2015,3]]},"DOI":"10.1109\/tac.2014.2357134","type":"journal-article","created":{"date-parts":[[2014,9,12]],"date-time":"2014-09-12T14:36:59Z","timestamp":1410532619000},"page":"743-758","source":"Crossref","is-referenced-by-count":13,"title":["A New Optimal Stepsize for Approximate Dynamic Programming"],"prefix":"10.1109","volume":"60","author":[{"given":"Ilya O.","family":"Ryzhov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peter I.","family":"Frazier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Warren B.","family":"Powell","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1609","article-title":"A convergent O(n) algorithm for off-policy temporal-difference learning with linear function approximation","volume":"21","author":"sutton","year":"2008","journal-title":"Adv Neural Inform Processing Syst"},{"key":"ref38","first-page":"705","article-title":"Temporal difference updating without a learning rate","volume":"20","author":"hutter","year":"2007","journal-title":"Advances in neural information processing systems"},{"key":"ref33","first-page":"171","article-title":"Adapting bias by gradient descent: An incremental version of delta-bar-delta","author":"sutton","year":"0","journal-title":"Proc Nat Conf Artif Intell"},{"key":"ref32","first-page":"343","article-title":"No more pesky learning rates","author":"schaul","year":"0","journal-title":"Proc of the 30th Int'l Conf on Machine Learning"},{"key":"ref31","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"J Machine Learning Res"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-75894-2"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-006-8365-9"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/s10626-006-8134-8"},{"journal-title":"Optimal Control and Estimation","year":"1994","author":"stengel","key":"ref35"},{"key":"ref34","first-page":"2121","article-title":"Tuning-free stepsize adaptation","author":"mahmood","year":"0","journal-title":"Proc IEEE Int Conf Acous Speech Signal Processing"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.1110.0470"},{"key":"ref40","first-page":"924","article-title":"Concurrent reinforcement learning from customer interactions","author":"silver","year":"0","journal-title":"Proc of the 30th Int'l Conf on Machine Learning"},{"journal-title":"Dynamic Probabilistic Systems Volume II SemiMarkov and Decision Processes","year":"1971","author":"howard","key":"ref11"},{"key":"ref12","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","author":"puterman","year":"1994","journal-title":"Markov Decision Processes"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.2307\/2002797"},{"journal-title":"Neuro-Dynamic Programming","year":"1996","author":"bertsekas","key":"ref14"},{"journal-title":"Reinforcement Learning","year":"1998","author":"sutton","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/9780470544785"},{"key":"ref17","doi-asserted-by":"crossref","DOI":"10.1002\/9781118029176","author":"powell","year":"2011","journal-title":"Approximate Dynamic Programming Solving the Curses of Dimensionality (2nd ed )"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.1997.652501"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/WSC.2008.4736109"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1287\/opre.1090.0768"},{"key":"ref27","first-page":"1","article-title":"Learning rates for Q-learning","volume":"5","author":"even-dar","year":"2003","journal-title":"J Machine Learning Res"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.ejor.2012.03.049"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.1090.1049"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1108\/17410380910929592"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/s12667-009-0007-4"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.1997.657615"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1287\/trsc.1090.0262"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.1090.0345"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1287\/trsc.1080.0238"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.1100.0433"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.1100.1143"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.1040.0079"},{"key":"ref45","first-page":"2079","article-title":"Value function approximation using multiple aggregation for multiattribute resource management","volume":"9","author":"george","year":"2008","journal-title":"J Machine Learning Res"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4899-2696-8"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1016\/j.ejor.2005.01.022"},{"journal-title":"Stochastic Approximation","year":"1969","author":"wasan","key":"ref21"},{"key":"ref42","doi-asserted-by":"crossref","DOI":"10.1007\/978-0-387-21606-5","author":"hastie","year":"2001","journal-title":"The Elements of Statistical Learning"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1994.6.6.1185"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2014.2357134"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993306"},{"journal-title":"Mathematical Statistics Basic Ideas and Selected Topics","year":"2001","author":"bickel","key":"ref44"},{"key":"ref26","first-page":"1064","article-title":"The asymptotic convergence-rate of Q-learning","volume":"10","author":"szepesv\u00e1ri","year":"1997","journal-title":"Advances in neural information processing systems"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1287\/opre.1110.0970"},{"key":"ref25","first-page":"2411","article-title":"Speedy Q-learning","volume":"24","author":"azar","year":"2011","journal-title":"Adv Neural Inform Processing Syst"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9\/7045467\/06897935.pdf?arnumber=6897935","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T11:40:58Z","timestamp":1641987658000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/6897935\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,3]]},"references-count":47,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tac.2014.2357134","relation":{},"ISSN":["0018-9286","1558-2523"],"issn-type":[{"type":"print","value":"0018-9286"},{"type":"electronic","value":"1558-2523"}],"subject":[],"published":{"date-parts":[[2015,3]]}}}