{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T10:08:08Z","timestamp":1740132488814,"version":"3.37.3"},"reference-count":23,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2023,8,1]],"date-time":"2023-08-01T00:00:00Z","timestamp":1690848000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","award":["NRF-2021R1F1A1061613"],"award-info":[{"award-number":["NRF-2021R1F1A1061613"]}],"id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]},{"name":"BK21 FOUR from the Ministry of Education, Republic of Korea"},{"name":"Institute of Information communications Technology Planning Evaluation","award":["2022-0-00469"],"award-info":[{"award-number":["2022-0-00469"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Automat. Contr."],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1109\/tac.2022.3213763","type":"journal-article","created":{"date-parts":[[2022,10,11]],"date-time":"2022-10-11T19:28:20Z","timestamp":1665516500000},"page":"5006-5013","source":"Crossref","is-referenced-by-count":0,"title":["New Versions of Gradient Temporal-Difference Learning"],"prefix":"10.1109","volume":"68","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4962-8478","authenticated-orcid":false,"given":"Donghwan","family":"Lee","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering, KAIST, Daejeon, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1515-5836","authenticated-orcid":false,"given":"Han-Dong","family":"Lim","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering, KAIST, Daejeon, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6061-3202","authenticated-orcid":false,"given":"Jihoon","family":"Park","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering, KAIST, Daejeon, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7536-3477","authenticated-orcid":false,"given":"Okyong","family":"Choi","sequence":"additional","affiliation":[{"name":"Shinhan Bank Research Team, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2018.2851375"},{"key":"ref12","volume":"35","author":"kushner","year":"2003","journal-title":"Stochastic Approximation and Recursive Algorithms and Applications"},{"article-title":"New versions of gradient temporal difference learning","year":"2021","author":"lee","key":"ref23"},{"key":"ref15","first-page":"1","article-title":"A generalized projected Bellman error for off-policy value estimation in reinforcement learning","volume":"23","author":"patterson","year":"2022","journal-title":"J Mach Learn Res"},{"key":"ref14","first-page":"3524","article-title":"Gradient temporal-difference learning with regularized corrections","author":"ghiassian","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"article-title":"Stochastic primal-dual methods and sample complexity of reinforcement learning","year":"2016","author":"chen","key":"ref20"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3211395"},{"key":"ref22","first-page":"417","article-title":"Off-policy temporal-difference learning with function approximation","author":"precup","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref10","article-title":"Multi-agent reinforcement learning via double averaging primal-dual optimization","volume":"31","author":"wai","year":"2018","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2016.7798956"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00115009"},{"key":"ref17","volume":"434","author":"bhatnagar","year":"2012","journal-title":"Stochastic Recursive Algorithms for Optimization Simultaneous Perturbation Methods"},{"journal-title":"Nonlinear Systems","year":"2002","author":"khalil","key":"ref16"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s10957-009-9522-7"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1137\/S0363012997331639"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.23919\/ECC.2019.8795670"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441"},{"key":"ref9","article-title":"Fast multi-agent temporal-difference learning via homotopy stochastic primal-dual method","author":"ding","year":"2019","journal-title":"Optim Found Reinforcement Learn Workshop 33rd Conf Neural Inf Process Syst"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553501"},{"key":"ref3","first-page":"1609","article-title":"A convergent $O(n)$ temporal-difference algorithm for off-policy learning with linear function approximation","author":"sutton","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref6","first-page":"1125","article-title":"SBEED: Convergent reinforcement learning with nonlinear function approximation","author":"dai","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2014.2368731"}],"container-title":["IEEE Transactions on Automatic Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9\/10196077\/09916098.pdf?arnumber=9916098","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,14]],"date-time":"2023-08-14T18:08:58Z","timestamp":1692036538000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9916098\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8]]},"references-count":23,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tac.2022.3213763","relation":{},"ISSN":["0018-9286","1558-2523","2334-3303"],"issn-type":[{"type":"print","value":"0018-9286"},{"type":"electronic","value":"1558-2523"},{"type":"electronic","value":"2334-3303"}],"subject":[],"published":{"date-parts":[[2023,8]]}}}