{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,9,20]],"date-time":"2023-09-20T14:16:28Z","timestamp":1695219388933},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"18","license":[{"start":{"date-parts":[[2023,4,28]],"date-time":"2023-04-28T00:00:00Z","timestamp":1682640000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,4,28]],"date-time":"2023-04-28T00:00:00Z","timestamp":1682640000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772355"],"award-info":[{"award-number":["61772355"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702055"],"award-info":[{"award-number":["61702055"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1007\/s10489-023-04579-4","type":"journal-article","created":{"date-parts":[[2023,4,28]],"date-time":"2023-04-28T08:02:18Z","timestamp":1682668938000},"page":"20917-20937","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Temporal-difference emphasis learning with regularized correction for off-policy evaluation and control"],"prefix":"10.1007","volume":"53","author":[{"given":"Jiaqing","family":"Cao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiming","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shan","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,28]]},"reference":[{"key":"4579_CR1","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction, 2nd edn. MIT press, Cambridge","edition":"2nd edn."},{"key":"4579_CR2","doi-asserted-by":"crossref","unstructured":"Liang D, Deng H, Liu Y (2022) The treatment of sepsis: an episodic memory-assisted deep reinforcement learning approach. Appl Intell, pp 1\u201311","DOI":"10.1007\/s10489-022-04099-7"},{"issue":"3","key":"4579_CR3","doi-asserted-by":"publisher","first-page":"1936","DOI":"10.1109\/TCYB.2020.2991166","volume":"52","author":"V Narayanan","year":"2022","unstructured":"Narayanan V, Modares H, Jagannathan S, Lewis FL (2022) Event-Driven Off-Policy Reinforcement learning for control of interconnected systems. IEEE Trans Cybern 52(3):1936\u20131946","journal-title":"IEEE Trans Cybern"},{"issue":"5","key":"4579_CR4","doi-asserted-by":"publisher","first-page":"2223","DOI":"10.1109\/TNNLS.2020.3044196","volume":"33","author":"W Meng","year":"2022","unstructured":"Meng W, Zheng Q, Shi Y, Pan G (2022) An Off-Policy trust region policy optimization method with monotonic improvement guarantee for deep reinforcement learning. IEEE Trans Neural Netw Learn Syst 33(5):2223\u20132235","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"10","key":"4579_CR5","doi-asserted-by":"publisher","first-page":"7233","DOI":"10.1007\/s10489-021-02257-x","volume":"51","author":"R Xu","year":"2021","unstructured":"Xu R, Li M, Yang Z, Yang L, Qiao K, Shang Z (2021) Dynamic feature selection algorithm based on Q-learning mechanism. Appl Intell 51(10):7233\u20137244","journal-title":"Appl Intell"},{"key":"4579_CR6","doi-asserted-by":"publisher","first-page":"110076","DOI":"10.1016\/j.automatica.2021.110076","volume":"136","author":"J Li","year":"2022","unstructured":"Li J, Xiao Z, Fan J, Chai T, Lewis FL (2022) Off-policy Q-learning: Solving Nash equilibrium of multi-player games with network-induced delay and unmeasured state. Automatica 136:110076","journal-title":"Automatica"},{"key":"4579_CR7","unstructured":"Jaderberg M, Mnih V, Czarnecki WM, Schaul T, Leibo JZ, Silver D, Kavukcuoglu K (2017) Reinforcement learning with unsupervised auxiliary tasks. In: Proceedings of the 5th International conference on learning representations"},{"key":"4579_CR8","unstructured":"Zahavy T, Xu Z, Veeriah V, Hessel M, Oh J, van Hasselt H, Silver D, Singh S (2020) A self-tuning actor-critic algorithm. In: Advances in neural information processing systems, pp 20913\u201320924"},{"issue":"7540","key":"4579_CR9","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"4579_CR10","unstructured":"Espeholt L, Soyer H, Munos R, Simonyan K, Mnih V, Ward T, Doron Y, Firoiu V, Harley T, Dunning I, Legg S, Kavukcuoglu K (2018) IMPALA: Scalable distributed deep-RL with importance weighted actor-learner architectures. In: Proceedings of the 35th International conference on machine learning, pp 1406\u20131415"},{"key":"4579_CR11","unstructured":"Metelli AM, Russo A, Restelli M (2021) Subgaussian and differentiable importance sampling for Off-Policy evaluation and learning. In: Advances in neural information processing systems, pp 8119\u20138132"},{"issue":"167","key":"4579_CR12","first-page":"1","volume":"21","author":"N Kallus","year":"2020","unstructured":"Kallus N, Uehara M (2020) Double reinforcement learning for efficient Off-Policy evaluation in markov decision processes. J Mach Learn Res 21(167):1\u201363","journal-title":"J Mach Learn Res"},{"key":"4579_CR13","volume-title":"Markov decision processes: discrete stochastic dynamic programming","author":"ML Puterman","year":"2014","unstructured":"Puterman ML (2014) Markov decision processes: discrete stochastic dynamic programming. Wiley, Hoboken"},{"key":"4579_CR14","unstructured":"Xie T, Ma Y, Wang Y (2019) Towards optimal Off-Policy evaluation for reinforcement learning with marginalized importance sampling. In: Advances in neural information processing systems, pp 9665\u20139675"},{"key":"4579_CR15","unstructured":"Shen SP, Ma YJ, Gottesman O, Doshi-velez F (2021) State relevance for off-policy evaluation. In: Proceedings of the 38th International conference on machine learning, pp 9537\u20139546"},{"key":"4579_CR16","unstructured":"Zhang S, Liu B, Whiteson S (2020) GradientDICE: Rethinking generalized Offline estimation of stationary values. In: Proceedings of the 37th International conference on machine learning, pp 11194\u201311203"},{"key":"4579_CR17","unstructured":"Liu Y, Swaminathan A, Agarwal A, Brunskill E (2020) Off-Policy Policy gradient with stationary distribution correction. In: Uncertainty in artificial intelligence, pp 1180\u20131190"},{"key":"4579_CR18","unstructured":"Zhang R, Dai B, Li L, Schuurmans D (2020) GenDICE: Generalized Offline estimation of stationary values. In: Proceedings of the 8th International conference on learning representations"},{"issue":"5","key":"4579_CR19","doi-asserted-by":"publisher","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"JN Tsitsiklis","year":"1997","unstructured":"Tsitsiklis JN, Van Roy B (1997) An analysis of temporal-difference learning with function approximation. IEEE Trans Auto Control 42(5):674\u2013690","journal-title":"IEEE Trans Auto Control"},{"key":"4579_CR20","doi-asserted-by":"crossref","unstructured":"Baird L (1995) Residual algorithms: reinforcement learning with function approximation. In: Proceedings of the 12th International conference on machine learning, pp 30\u201337","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"key":"4579_CR21","unstructured":"Xu T, Yang Z, Wang Z, Liang Y (2021) Doubly robust Off-Policy actor-critic: convergence and optimality. In: Proceedings of the 38th International conference on machine learning, pp 11581\u201311591"},{"key":"4579_CR22","unstructured":"Degris T, White M, Sutton RS (2012) Off-policy actor-critic, arXiv:1205.4839"},{"issue":"1","key":"4579_CR23","first-page":"2603","volume":"17","author":"RS Sutton","year":"2016","unstructured":"Sutton RS, Mahmood AR, White M (2016) An emphatic approach to the problem of off-policy temporal-difference learning. J Mach Learn Res 17(1):2603\u20132631","journal-title":"J Mach Learn Res"},{"key":"4579_CR24","unstructured":"Imani E, Graves E, White M (2018) An Off-policy policy gradient theorem using emphatic weightings. In: Advances in neural information processing systems, pp 96\u2013106"},{"key":"4579_CR25","unstructured":"Zhang S, Liu B, Yao H, Whiteson S (2020) Provably convergent Two-Timescale Off-Policy Actor-Critic with function approximation. In: Proceedings of the 37th International Conference on Machine Learning, 11204\u201311213"},{"key":"4579_CR26","doi-asserted-by":"crossref","unstructured":"Sutton RS, Maei HR, Precup D, Bhatnagar S, Silver D, Szepesv\u00e1ri C, Wiewiora E (2009) Fast gradient-descent methods for temporal-difference learning with linear function approximation. In: Proceedings of the 26th International conference on machine learning, pp 993\u20131000","DOI":"10.1145\/1553374.1553501"},{"key":"4579_CR27","doi-asserted-by":"crossref","unstructured":"Sutton RS, Szepesv\u00e1ri C, Maei HR (2008) A convergent o(n) temporal-difference algorithm for off-policy learning with linear function approximation. In: Advances in neural information processing systems, pp 1609\u20131616","DOI":"10.1145\/1553374.1553501"},{"key":"4579_CR28","unstructured":"Maei HR (2011) Gradient temporal-difference learning algorithms, Phd thesis, University of Alberta"},{"key":"4579_CR29","unstructured":"Xu T, Zou S, Liang Y (2019) Two time-scale off-policy TD learning: Non-asymptotic analysis over Markovian samples. In: Advances in neural information processing systems, pp 10633\u201310643"},{"key":"4579_CR30","unstructured":"Ma S, Zhou Y, Zou S (2020) Variance-reduced off-policy TDC Learning: Non-asymptotic convergence analysis. In: Advances in neural information processing systems, pp 14796\u201314806"},{"key":"4579_CR31","unstructured":"Ghiassian S, Patterson A, Garg S, Gupta D, White A, White M (2020) Gradient temporal-difference learning with regularized corrections. In: Proceedings of the 37th International conference on machine learning, pp 3524\u20133534"},{"issue":"1","key":"4579_CR32","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/BF00115009","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton RS (1988) Learning to predict by the methods of temporal differences. Mach Learn 3 (1):9\u201344","journal-title":"Mach Learn"},{"key":"4579_CR33","unstructured":"Jiang R, Zahavy T, Xu Z, White A, Hessel M, Blundell C, van Hasselt H (2021) Emphatic algorithms for deep reinforcement learning. In: Proceedings of the 38th International conference on machine learning, pp 5023\u20135033"},{"key":"4579_CR34","doi-asserted-by":"crossref","unstructured":"Jiang R, Zhang S, Chelu V, White A, van Hasselt H (2022) Learning expected emphatic traces for deep RL. In: Proceedings of the 36th AAAI Conference on artificial intelligence, pp 12882\u201312890","DOI":"10.1609\/aaai.v36i6.20660"},{"key":"4579_CR35","unstructured":"Guan Z, Xu T, Liang Y (2022) PER-ETD: A polynomially efficient emphatic temporal difference learning method. In: 10Th international conference on learning representations"},{"key":"4579_CR36","unstructured":"Yu H (2015) On convergence of emphatic Temporal-Difference learning. In: Proceedings of The 28th Conference on learning theory, pp 1724\u20131751"},{"issue":"1","key":"4579_CR37","first-page":"7745","volume":"17","author":"H Yu","year":"2016","unstructured":"Yu H (2016) Weak convergence properties of constrained emphatic temporal-difference learning with constant and slowly diminishing stepsize. J Mach Learn Res 17(1):7745\u20137802","journal-title":"J Mach Learn Res"},{"key":"4579_CR38","unstructured":"Ghiassian S, Rafiee B, Sutton RS (2016) A first empirical study of emphatic temporal difference learning. In: Advances in neural information processing systems"},{"key":"4579_CR39","unstructured":"Gu X, Ghiassian S, Sutton RS (2019) Should all temporal difference learning use emphasis? arXiv:1903.00194"},{"key":"4579_CR40","doi-asserted-by":"crossref","unstructured":"Hallak A, Tamar A, Munos R, Mannor S (2016) Generalized emphatic temporal difference learning: Bias-Variance analysis. In: Proceedings of the 30th AAAI Conference on artificial intelligence, pp 1631\u20131637","DOI":"10.1609\/aaai.v30i1.10227"},{"key":"4579_CR41","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1016\/j.ins.2021.08.082","volume":"580","author":"J Cao","year":"2021","unstructured":"Cao J, Liu Q, Zhu F, Fu Q, Zhong S (2021) Gradient temporal-difference learning for off-policy evaluation using emphatic weightings. Inf Sci 580:311\u2013330","journal-title":"Inf Sci"},{"issue":"153","key":"4579_CR42","first-page":"1","volume":"23","author":"S Zhang","year":"2022","unstructured":"Zhang S, Whiteson S (2022) Truncated emphatic temporal difference methods for prediction and control. J Mach Learn Res 23(153):1\u201359","journal-title":"J Mach Learn Res"},{"key":"4579_CR43","doi-asserted-by":"crossref","unstructured":"van Hasselt H, Madjiheurem S, Hessel M, Silver D, Barreto A, Borsa D (2021) Expected eligibility traces. In: Proceedings of the 35th AAAI Conference on artificial intelligence, pp 9997\u201310005","DOI":"10.1609\/aaai.v35i11.17200"},{"key":"4579_CR44","unstructured":"Hallak A, Mannor S (2017) Consistent on-line off-policy evaluation. In: Proceedings of the 34th International conference on machine learning, pp 1372\u20131383"},{"key":"4579_CR45","unstructured":"Liu Q, Li L, Tang Z, Zhou D (2018) Breaking the curse of horizon: Infinite-horizon off-policy estimation. In: Advances in neural information processing systems, pp 5361\u20135371"},{"key":"4579_CR46","doi-asserted-by":"crossref","unstructured":"Gelada C, Bellemare MG (2019) Off-Policy Deep reinforcement learning by bootstrapping the covariate shift. In: Proceedings of the 33th AAAI Conference on artificial intelligence, pp 3647\u20133655","DOI":"10.1609\/aaai.v33i01.33013647"},{"key":"4579_CR47","unstructured":"Nachum O, Chow Y, Dai B, Li L (2019) DualDICE: Behavior-Agnostic estimation of discounted stationary distribution corrections. In: Advances in neural information processing systems, pp 2315\u20132325"},{"key":"4579_CR48","unstructured":"Zhang S, Yao H, Whiteson S (2021) Breaking the deadly triad with a target network. In: Proceedings of the 38th International conference on machine learning, pp 12621\u201312631"},{"key":"4579_CR49","unstructured":"Zhang S, Wan Y, Sutton RS, Whiteson S (2021) Average-Reward Off-Policy Policy evaluation with function approximation. In: Proceedings of the 38th International conference on machine learning, pp 12578\u201312588"},{"key":"4579_CR50","unstructured":"Wang T, Bowling M, Schuurmans D, Lizotte DJ (2008) Stable dual dynamic programming. In: Advances in neural information processing systems, pp 1569\u20131576"},{"key":"4579_CR51","unstructured":"Hallak A, Mannor S (2017) Consistent on-line off-policy evaluation. In: Proceedings of the 34th International conference on machine learning, pp 1372\u20131383"},{"key":"4579_CR52","unstructured":"Zhang S, Veeriah V, Whiteson S (2020) Learning retrospective knowledge with reverse reinforcement learning. In: Advances in neural information processing systems, pp 19976\u2013 19987"},{"key":"4579_CR53","unstructured":"Precup D, Sutton RS, Dasgupta S (2001) Off-Policy Temporal difference learning with function approximation. In: Proceedings of the 18th International conference on machine learning, pp 417\u2013424"},{"key":"4579_CR54","unstructured":"Zhang S, Boehmer W, Whiteson S (2019) Generalized off-policy actor-critic. In: Advances in neural information processing systems, pp 1999\u20132009"},{"key":"4579_CR55","doi-asserted-by":"crossref","unstructured":"Robbins H, Monro S (1951) A stochastic approximation method. Ann Math Stat, pp 400\u2013407","DOI":"10.1214\/aoms\/1177729586"},{"key":"4579_CR56","doi-asserted-by":"crossref","unstructured":"Levin DA, Peres Y (2017) Markov chains and mixing times, vol 107, American Mathematical Soc.","DOI":"10.1090\/mbk\/107"},{"key":"4579_CR57","unstructured":"Kolter JZ (2011) The fixed points of Off-Policy TD. In: Advances in neural information processing systems, pp 2169\u20132177"},{"key":"4579_CR58","unstructured":"White A, White M (2016) Investigating practical linear temporal difference learning. In: Proceedings of the 15th International conference on autonomous agents & multiagent systems, pp 494\u2013502"},{"key":"4579_CR59","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym, arXiv:1606.01540"},{"issue":"2","key":"4579_CR60","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1137\/S0363012997331639","volume":"38","author":"VS Borkar","year":"2000","unstructured":"Borkar VS, Meyn SP (2000) The ODE method for convergence of stochastic approximation and reinforcement learning. SIAM J Control Optim 38(2):447\u2013469","journal-title":"SIAM J Control Optim"},{"key":"4579_CR61","doi-asserted-by":"crossref","unstructured":"Golub GH, Van Loan CF (2013) Matrix computations, vol 3, Johns Hopkins University Press","DOI":"10.56021\/9781421407944"},{"key":"4579_CR62","unstructured":"White M (2017) Unifying task specification in reinforcement learning. In: Proceedings of the 34th International conference on machine learning, pp 3742\u20133750"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-04579-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-023-04579-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-023-04579-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,19]],"date-time":"2023-09-19T11:09:07Z","timestamp":1695121747000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-023-04579-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,28]]},"references-count":62,"journal-issue":{"issue":"18","published-print":{"date-parts":[[2023,9]]}},"alternative-id":["4579"],"URL":"https:\/\/doi.org\/10.1007\/s10489-023-04579-4","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,4,28]]},"assertion":[{"value":"17 March 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 April 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Ethics approval"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Consent to Participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Consent for Publication"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Competing interests"}}]}}