{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,23]],"date-time":"2025-12-23T10:44:32Z","timestamp":1766486672268,"version":"3.28.0"},"reference-count":35,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,6]]},"DOI":"10.23919\/acc.2018.8430925","type":"proceedings-article","created":{"date-parts":[[2018,8,17]],"date-time":"2018-08-17T20:16:10Z","timestamp":1534536970000},"page":"6608-6615","source":"Crossref","is-referenced-by-count":16,"title":["Nonparametric Stochastic Compositional Gradient Descent for Q-Learning in Continuous Markov Decision Problems"],"prefix":"10.23919","author":[{"given":"Ekaterina","family":"Tolstaya","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alec","family":"Koppel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ethan","family":"Stump","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alejandro","family":"Ribeiro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","first-page":"785","article-title":"Online learning with kernels","author":"kivinen","year":"2002","journal-title":"Advances in neural information processing systems"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1137\/S0363012997331639"},{"key":"ref31","first-page":"2651","article-title":"Universal kernels","volume":"7","author":"micchelli","year":"2006","journal-title":"J Mach Learn Res"},{"key":"ref30","first-page":"2507","article-title":"When is there a representer theorem? vector versus matrix regularizers","volume":"10","author":"argyriou","year":"2009","journal-title":"J Mach Learn Res"},{"year":"0","journal-title":"Openai gym - continuous mountain car","key":"ref35"},{"year":"2015","author":"lillicrap","journal-title":"Continuous control with deep reinforcement learning","key":"ref34"},{"year":"1957","author":"bellman","journal-title":"Dynamic Programming","key":"ref10"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1007\/BF00993306"},{"key":"ref12","first-page":"1609","article-title":"A convergent o temporal-difference algorithm for off-policy learning with linear function approximation","author":"sutton","year":"2009","journal-title":"Advances in neural information processing systems"},{"key":"ref13","first-page":"1204","article-title":"Convergent temporal-difference learning with arbitrary smooth function approximation","author":"bhatnagar","year":"2009","journal-title":"Advances in neural information processing systems"},{"year":"2013","author":"mnih","journal-title":"Playing atari with deep reinforcement learning","key":"ref14"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.1016\/0022-247X(71)90184-3"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"416","DOI":"10.1007\/3-540-44581-1_27","article-title":"A generalized representer theorem","author":"sch\u00f6lkopf","year":"2001","journal-title":"Conf Computational Learning Theory"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/9.580874"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1145\/1329125.1329242"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1007\/s10107-016-1017-3"},{"key":"ref4","volume":"23","author":"bertsekas","year":"1978","journal-title":"Stochastic Optimal Control The Discrete Time Case"},{"key":"ref27","article-title":"Bayes meets bellman: The gaussian process approach to temporal difference learning","author":"engel","year":"2003","journal-title":"Proc of the 20th Int Conf on Machine Learning"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1177\/0278364913495721"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.5937\/jemc1402059M"},{"key":"ref29","article-title":"Efficient memory-based learning for robot control","author":"moore","year":"1990","journal-title":"University of Cambridge Computer Laboratory Tech Rep UCAM-CL-TR-725"},{"year":"2018","author":"sutton","journal-title":"Reinforcement Learning An Introduction","key":"ref5"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1007\/BF00115009"},{"key":"ref7","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","author":"sutton","year":"2000","journal-title":"Advances in neural information processing systems"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1090\/S0002-9904-1954-09848-8"},{"year":"1989","author":"watkins","journal-title":"Learning from delayed rewards","key":"ref9"},{"key":"ref1","article-title":"Nonparametric stochastic compositional gradient descent for q-learning in continuous markov decision problems","author":"koppel","year":"2017","journal-title":"IEEE Trans Autom Control (under preparation)"},{"key":"ref20","doi-asserted-by":"crossref","DOI":"10.1137\/1.9781611973433","volume":"16","author":"shapiro","year":"2014","journal-title":"Lectures on Stochastic Programming Modeling and Theory SIAM"},{"year":"2017","author":"koppel","journal-title":"Breaking Bellman's Curse of Dimensionality Efficient Kernel Gradient Temporal Difference","key":"ref22"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1080\/17442508308833246"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.1109\/78.258082"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1023\/A:1017928328829"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1016\/j.crma.2008.03.014"},{"year":"2016","author":"koppel","journal-title":"Parsimonious online learning with kernels via sparse projections in function space","key":"ref25"}],"event":{"name":"2018 Annual American Control Conference (ACC)","start":{"date-parts":[[2018,6,27]]},"location":"Milwaukee, WI","end":{"date-parts":[[2018,6,29]]}},"container-title":["2018 Annual American Control Conference (ACC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8410068\/8430677\/08430925.pdf?arnumber=8430925","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,8,24]],"date-time":"2020-08-24T02:56:48Z","timestamp":1598237808000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8430925\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,6]]},"references-count":35,"URL":"https:\/\/doi.org\/10.23919\/acc.2018.8430925","relation":{},"subject":[],"published":{"date-parts":[[2018,6]]}}}