{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T03:20:49Z","timestamp":1740108049123,"version":"3.37.3"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"32","license":[{"start":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T00:00:00Z","timestamp":1694390400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T00:00:00Z","timestamp":1694390400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772355"],"award-info":[{"award-number":["61772355"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702055"],"award-info":[{"award-number":["61702055"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1007\/s00521-023-08965-4","type":"journal-article","created":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T08:02:05Z","timestamp":1694419325000},"page":"23599-23616","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Generalized gradient emphasis learning for off-policy evaluation and control with function approximation"],"prefix":"10.1007","volume":"35","author":[{"given":"Jiaqing","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8710-1810","authenticated-orcid":false,"given":"Quan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiming","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shan","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,11]]},"reference":[{"key":"8965_CR1","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction, 2nd edn. MIT press, Cambridge","edition":"2"},{"key":"8965_CR2","doi-asserted-by":"publisher","first-page":"5255","DOI":"10.1007\/s00521-021-06476-8","volume":"34","author":"M Mohammadi","year":"2022","unstructured":"Mohammadi M, Arefi MM, Vafamand N, Kaynak O (2022) Control of an AUV with completely unknown dynamics and multi-asymmetric input constraints via off-policy reinforcement learning. Neural Comput Appl 34:5255\u20135265","journal-title":"Neural Comput Appl"},{"key":"8965_CR3","doi-asserted-by":"publisher","first-page":"1936","DOI":"10.1109\/TCYB.2020.2991166","volume":"52","author":"V Narayanan","year":"2022","unstructured":"Narayanan V, Modares H, Jagannathan S, Lewis FL (2022) Event-driven off-policy reinforcement learning for control of interconnected systems. IEEE Trans Cybern 52:1936\u20131946","journal-title":"IEEE Trans Cybern"},{"key":"8965_CR4","doi-asserted-by":"publisher","first-page":"2223","DOI":"10.1109\/TNNLS.2020.3044196","volume":"33","author":"W Meng","year":"2022","unstructured":"Meng W, Zheng Q, Shi Y, Pan G (2022) An off-policy trust region policy optimization method with monotonic improvement guarantee for deep reinforcement learning. IEEE Trans Neural Netw Learn Syst 33:2223\u20132235","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"8965_CR5","doi-asserted-by":"publisher","first-page":"14431","DOI":"10.1007\/s00521-019-04556-4","volume":"32","author":"H Kong","year":"2020","unstructured":"Kong H, Yan J, Wang H, Fan L (2020) Energy management strategy for electric vehicles based on deep q-learning using Bayesian optimization. Neural Comput Appl 32:14431\u201314445","journal-title":"Neural Comput Appl"},{"key":"8965_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2021.110076","volume":"136","author":"J Li","year":"2022","unstructured":"Li J, Xiao Z, Fan J, Chai T, Lewis FL (2022) Off-policy q-learning: Solving nash equilibrium of multi-player games with network-induced delay and unmeasured state. Automatica 136:110076","journal-title":"Automatica"},{"key":"8965_CR7","unstructured":"Jaderberg M, Mnih V, Czarnecki WM, Schaul T, Leibo JZ, Silver D, Kavukcuoglu K (2017) Reinforcement learning with unsupervised auxiliary tasks. In: Proceedings of the 5th international conference on learning representations"},{"key":"8965_CR8","unstructured":"Zahavy T, Xu Z, Veeriah V, Hessel M, Oh J, van Hasselt H, Silver D, Singh S (2020) A self-tuning actor-critic algorithm. In: Advances in neural information processing systems, pp 20913\u201320924"},{"key":"8965_CR9","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518:529\u2013533","journal-title":"Nature"},{"key":"8965_CR10","unstructured":"Espeholt L, Soyer H, Munos R, Simonyan K, Mnih V, Ward T, Doron Y, Firoiu V, Harley T, Dunning I, Legg S, Kavukcuoglu K (2018) IMPALA: scalable distributed deep-RL with importance weighted actor-learner architectures. In: Proceedings of the 35th international conference on machine learning, pp 1406\u20131415"},{"key":"8965_CR11","unstructured":"Jiang R, Zahavy T, Xu Z, White A, Hessel M, Blundell C, van Hasselt H (2021) Emphatic algorithms for deep reinforcement learning. In: Proceedings of the 38th international conference on machine learning, pp 5023\u20135033"},{"key":"8965_CR12","doi-asserted-by":"crossref","unstructured":"Jiang R, Zhang S, Chelu V, White A, van Hasselt H (2022) Learning expected emphatic traces for deep RL. In: Proceedings of the 36th AAAI conference on artificial intelligence, pp 12882\u201312890","DOI":"10.1609\/aaai.v36i6.20660"},{"key":"8965_CR13","unstructured":"Guan Z, Xu T, Liang Y (2022) PER-ETD: a polynomially efficient emphatic temporal difference learning method. In: 10th International conference on learning representations"},{"key":"8965_CR14","unstructured":"Zhang S, Liu B, Whiteson S (2020) Gradientdice: rethinking generalized offline estimation of stationary values. In: Proceedings of the 37th international conference on machine learning, pp 11194\u201311203"},{"key":"8965_CR15","unstructured":"Liu Y, Swaminathan A, Agarwal A, Brunskill E (2020) Off-policy policy gradient with stationary distribution correction. In: Uncertainty in artificial intelligence, pp 1180\u20131190"},{"key":"8965_CR16","unstructured":"Zhang R, Dai B, Li L, Schuurmans D (2020) Gendice: generalized offline estimation of stationary values. In: Proceedings of the 8th international conference on learning representations"},{"key":"8965_CR17","unstructured":"Metelli AM, Russo A, Restelli M (2021) Subgaussian and differentiable importance sampling for off-policy evaluation and learning. In: Advances in neural information processing systems, pp 8119\u20138132"},{"key":"8965_CR18","first-page":"1","volume":"21","author":"N Kallus","year":"2020","unstructured":"Kallus N, Uehara M (2020) Double reinforcement learning for efficient off-policy evaluation in Markov decision processes. J Mach Learn Res 21:1\u201363","journal-title":"J Mach Learn Res"},{"key":"8965_CR19","volume-title":"Markov decision processes: discrete stochastic dynamic programming","author":"ML Puterman","year":"2014","unstructured":"Puterman ML (2014) Markov decision processes: discrete stochastic dynamic programming. John Wiley & Sons, New York"},{"key":"8965_CR20","unstructured":"Wai H, Hong M, Yang Z, Wang Z, Tang K (2019) Variance reduced policy evaluation with smooth function approximation. In: Advances in neural information processing systems, pp 5776\u20135787"},{"key":"8965_CR21","unstructured":"Shen SP, Ma YJ, Gottesman O, Doshi-Velez F (2021) State relevance for off-policy evaluation. In: Proceedings of the 38th international conference on machine learning, pp 9537\u20139546"},{"key":"8965_CR22","unstructured":"Degris T, White M, Sutton RS (2012) Off-policy actor-critic. arXiv preprint arXiv:1205.4839"},{"key":"8965_CR23","first-page":"2603","volume":"17","author":"RS Sutton","year":"2016","unstructured":"Sutton RS, Mahmood AR, White M (2016) An emphatic approach to the problem of off-policy temporal-difference learning. J Mach Learn Res 17:2603\u20132631","journal-title":"J Mach Learn Res"},{"key":"8965_CR24","unstructured":"Imani E, Graves E, White M (2018) An off-policy policy gradient theorem using emphatic weightings. In: Advances in neural information processing systems, pp 96\u2013106"},{"key":"8965_CR25","unstructured":"Zhang S, Liu B, Yao H, Whiteson S (2020) Provably convergent two-timescale off-policy actor-critic with function approximation. In: Proceedings of the 37th international conference on machine learning, pp 11204\u201311213"},{"key":"8965_CR26","doi-asserted-by":"crossref","unstructured":"Hallak A, Tamar A, Munos R, Mannor S (2016) Generalized emphatic temporal difference learning: Bias-variance analysis. In: Proceedings of the 30th AAAI conference on artificial intelligence, pp 1631\u20131637","DOI":"10.1609\/aaai.v30i1.10227"},{"key":"8965_CR27","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/BF00115009","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton RS (1988) Learning to predict by the methods of temporal differences. Mach Learn 3:9\u201344","journal-title":"Mach Learn"},{"key":"8965_CR28","doi-asserted-by":"publisher","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"JN Tsitsiklis","year":"1997","unstructured":"Tsitsiklis JN, Van Roy B (1997) An analysis of temporal-difference learning with function approximation. IEEE Trans Autom Control 42:674\u2013690","journal-title":"IEEE Trans Autom Control"},{"key":"8965_CR29","doi-asserted-by":"crossref","unstructured":"Baird L (1995) Residual algorithms: reinforcement learning with function approximation. In: Proceedings of the 12th international conference on machine learning, pp 30\u201337","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"key":"8965_CR30","doi-asserted-by":"crossref","unstructured":"Sutton RS, Szepesv\u00e1ri C, Maei HR (2008) A convergent o(n) temporal-difference algorithm for off-policy learning with linear function approximation. In: Advances in neural information processing systems, pp 1609\u20131616","DOI":"10.1145\/1553374.1553501"},{"key":"8965_CR31","doi-asserted-by":"crossref","unstructured":"Sutton RS, Maei HR, Precup D, Bhatnagar S, Silver D, Szepesv\u00e1ri C, Wiewiora E (2009) Fast gradient-descent methods for temporal-difference learning with linear function approximation. In: Proceedings of the 26th international conference on machine learning, pp 993\u20131000","DOI":"10.1145\/1553374.1553501"},{"key":"8965_CR32","unstructured":"Maei HR (2011) Gradient temporal-difference learning algorithms, Phd thesis, University of Alberta"},{"key":"8965_CR33","first-page":"1","volume":"23","author":"S Zhang","year":"2022","unstructured":"Zhang S, Whiteson S (2022) Truncated emphatic temporal difference methods for prediction and control. J Mach Learn Res 23:1\u201359","journal-title":"J Mach Learn Res"},{"key":"8965_CR34","doi-asserted-by":"crossref","unstructured":"van Hasselt H, Madjiheurem S, Hessel M, Silver D, Barreto A, Borsa D (2021) Expected eligibility traces. In: Proceedings of the 35th AAAI conference on artificial intelligence, pp 9997\u201310005","DOI":"10.1609\/aaai.v35i11.17200"},{"key":"8965_CR35","unstructured":"Hallak A, Mannor S (2017) Consistent on-line off-policy evaluation. In: Proceedings of the 34th international conference on machine learning, pp 1372\u20131383"},{"key":"8965_CR36","unstructured":"Liu Q, Li L, Tang Z, Zhou D (2018) Breaking the curse of horizon: infinite-horizon off-policy estimation. In: Advances in neural information processing systems, pp 5361\u20135371"},{"key":"8965_CR37","doi-asserted-by":"crossref","unstructured":"Gelada C, Bellemare MG (2019) Off-policy deep reinforcement learning by bootstrapping the covariate shift. In: Proceedings of the 33th AAAI conference on artificial intelligence, pp 3647\u20133655","DOI":"10.1609\/aaai.v33i01.33013647"},{"key":"8965_CR38","unstructured":"Nachum O, Chow Y, Dai B, Li L (2019) Dualdice: behavior-agnostic estimation of discounted stationary distribution corrections. In: Advances in neural information processing systems, pp 2315\u20132325"},{"key":"8965_CR39","unstructured":"Zhang S, Yao H, Whiteson S (2021a) Breaking the deadly triad with a target network. In: Proceedings of the 38th international conference on machine learning, pp 12621\u201312631"},{"key":"8965_CR40","unstructured":"Zhang S, Wan Y, Sutton RS, Whiteson S (2021b) Average-reward off-policy policy evaluation with function approximation. In: Proceedings of the 38th international conference on machine learning, pp 12578\u201312588"},{"key":"8965_CR41","doi-asserted-by":"crossref","unstructured":"Wang T, Bowling M, Schuurmans D (2007) Dual representations for dynamic programming and reinforcement learning. In: 2007 IEEE International symposium on approximate dynamic programming and reinforcement learning, pp 44\u201351","DOI":"10.1109\/ADPRL.2007.368168"},{"key":"8965_CR42","unstructured":"Wang T, Bowling M, Schuurmans D, Lizotte DJ (2008) Stable dual dynamic programming. In: Advances in neural information processing systems, pp 1569\u20131576"},{"key":"8965_CR43","unstructured":"Hallak A, Mannor S (2017) Consistent on-line off-policy evaluation. In: Proceedings of the 34th international conference on machine learning, pp 1372\u20131383"},{"key":"8965_CR44","unstructured":"Zhang S, Veeriah V, Whiteson S (2020) Learning retrospective knowledge with reverse reinforcement learning. In: Advances in neural information processing systems, pp 19976\u201319987"},{"key":"8965_CR45","unstructured":"Precup D, Sutton RS, Dasgupta S (2001) Off-policy temporal difference learning with function approximation. In: Proceedings of the 18th international conference on machine learning, pp 417\u2013424"},{"key":"8965_CR46","unstructured":"Zhang S, Boehmer W, Whiteson S (2019) Generalized off-policy actor-critic. In: Advances in neural information processing systems, pp 1999\u20132009"},{"key":"8965_CR47","doi-asserted-by":"crossref","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method, The annals of mathematical statistics, pp 400\u2013407","DOI":"10.1214\/aoms\/1177729586"},{"key":"8965_CR48","unstructured":"Yu H (2015) On convergence of emphatic temporal-difference learning. In: Proceedings of the 28th conference on learning theory, pp 1724\u20131751"},{"key":"8965_CR49","first-page":"7745","volume":"17","author":"H Yu","year":"2016","unstructured":"Yu H (2016) Weak convergence properties of constrained emphatic temporal-difference learning with constant and slowly diminishing stepsize. J Mach Learn Res 17:7745\u20137802","journal-title":"J Mach Learn Res"},{"key":"8965_CR50","doi-asserted-by":"crossref","unstructured":"Levin DA, Peres Y (2017) Markov chains and mixing times, vol 107. American Mathematical Soc","DOI":"10.1090\/mbk\/107"},{"key":"8965_CR51","unstructured":"Ghiassian S, Patterson A, Garg S, Gupta D, White A, White M (2020) Gradient temporal-difference learning with regularized corrections. In: Proceedings of the 37th international conference on machine learning, pp 3524\u20133534"},{"key":"8965_CR52","unstructured":"Bertsekas D, Tsitsiklis J (1989) Parallel and distributed computation: numeral methods"},{"key":"8965_CR53","unstructured":"Kolter JZ (2011) The fixed points of off-policy TD. In: Advances in neural information processing systems, pp 2169\u20132177"},{"key":"8965_CR54","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym. arXiv preprint arXiv:1606.01540"},{"key":"8965_CR55","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9781139020411","volume-title":"Matrix analysis","author":"RA Horn","year":"2012","unstructured":"Horn RA, Johnson CR (2012) Matrix analysis, 2nd edn. Cambridge University Press, Cambridge","edition":"2"},{"key":"8965_CR56","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1137\/S0363012997331639","volume":"38","author":"VS Borkar","year":"2000","unstructured":"Borkar VS, Meyn SP (2000) The ode method for convergence of stochastic approximation and reinforcement learning. SIAM J Control Optim 38:447\u2013469","journal-title":"SIAM J Control Optim"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08965-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-023-08965-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-023-08965-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,17]],"date-time":"2023-10-17T18:18:07Z","timestamp":1697566687000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-023-08965-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,11]]},"references-count":56,"journal-issue":{"issue":"32","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["8965"],"URL":"https:\/\/doi.org\/10.1007\/s00521-023-08965-4","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2023,9,11]]},"assertion":[{"value":"1 October 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 August 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 September 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}