{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T14:38:08Z","timestamp":1742913488450,"version":"3.40.3"},"publisher-location":"Cham","reference-count":23,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030286187"},{"type":"electronic","value":"9783030286194"}],"license":[{"start":{"date-parts":[[2019,11,28]],"date-time":"2019-11-28T00:00:00Z","timestamp":1574899200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-28619-4_31","type":"book-chapter","created":{"date-parts":[[2019,11,28]],"date-time":"2019-11-28T00:04:15Z","timestamp":1574899455000},"page":"387-404","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reachability and Differential Based Heuristics for Solving Markov Decision Processes"],"prefix":"10.1007","author":[{"given":"Shoubhik","family":"Debnath","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lantao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaurav","family":"Sukhatme","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,11,28]]},"reference":[{"key":"31_CR1","unstructured":"Andre, D., Friedman, N., Parr, R.: Generalized prioritized sweeping. Advances in Neural Information Processing Systems (1998)"},{"key":"31_CR2","unstructured":"Andre, D., Russell, S.J.: State abstraction for programmable reinforcement learning agents. In: AAAI\/IAAI, pp. 119\u2013125 (2002)"},{"issue":"1","key":"31_CR3","doi-asserted-by":"publisher","first-page":"185","DOI":"10.2307\/3213758","volume":"22","author":"D Assaf","year":"1985","unstructured":"Assaf, D., Shared, M., Shanthikumar, J.G.: First-passage times with PFr densities. J. Appl. Probab. 22(1), 185\u2013196 (1985)","journal-title":"J. Appl. Probab."},{"issue":"1\u20132","key":"31_CR4","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"AG Barto","year":"1995","unstructured":"Barto, A.G., Bradtke, S.J., Singh, S.P.: Learning to act using real-time dynamic programming. Artif. Intell. 72(1\u20132), 81\u2013138 (1995)","journal-title":"Artif. Intell."},{"key":"31_CR5","unstructured":"Bertsekas, D.P.: Dynamic Programming: Deterministic and Stochastic Models. Prentice-Hall, Englewood Cliffs (1987)"},{"key":"31_CR6","first-page":"12","volume":"3","author":"B Bonet","year":"2003","unstructured":"Bonet, B., Geffner, H.: Labeled RTDP: improving the convergence of real-time dynamic programming. ICAPS 3, 12\u201321 (2003)","journal-title":"ICAPS"},{"key":"31_CR7","unstructured":"Boutilier, C., Brafman, R.I., Geib, C.: Structured reachability analysis for Markov decision processes. In: Proceedings of the Fourteenth Conference on Uncertainty in Artificial Intelligence, pp. 24\u201332. Morgan Kaufmann Publishers Inc., (1998)"},{"key":"31_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1613\/jair.575","volume":"11","author":"C Boutilier","year":"1999","unstructured":"Boutilier, C., Dean, T., Hanks, S.: Decision-theoretic planning: structural assumptions and computational leverage. J. Artif. Intell. Res. 11, 1\u201394 (1999)","journal-title":"J. Artif. Intell. Res."},{"issue":"1","key":"31_CR9","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1016\/S0004-3702(00)00033-3","volume":"121","author":"C Boutilier","year":"2000","unstructured":"Boutilier, C., Dearden, R., Goldszmidt, M.: Stochastic dynamic programming with factored representations. Artif. Intell. 121(1), 49\u2013107 (2000)","journal-title":"Artif. Intell."},{"key":"31_CR10","unstructured":"Busoniu, L., Babuska, R., De Schutter, B., Ernst, D.: Reinforcement Learning and Dynamic Programming Using Function Approximators, vol. 39. CRC Press, Boca Raton (2010)"},{"key":"31_CR11","unstructured":"Golub, G.H., Van Loan, C.F.: Matrix Computations, 3rd edn. Johns Hopkins University Press, Baltimore (1996)"},{"issue":"1\u20132","key":"31_CR12","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1016\/S0004-3702(01)00106-0","volume":"129","author":"EA Hansen","year":"2001","unstructured":"Hansen, E.A., Zilberstein, S.: Lao: a heuristic search algorithm that finds solutions with loops. Artif. Intell. 129(1\u20132), 35\u201362 (2001)","journal-title":"Artif. Intell."},{"key":"31_CR13","unstructured":"Howard, R.A.: Dynamic Programming and Markov Processes. MIT Press, Cambridge (1960)"},{"key":"31_CR14","unstructured":"Kemeny, J.G., Mirkill, H., Snell, J.L., Thompson, G.L.: Finite Mathematical Structures. Prentice-Hall, Upper Saddle River (1959)"},{"key":"31_CR15","unstructured":"Li, L., Walsh, T.J., Littman, M.L.: Towards a unified theory of state abstraction for MDPs. In: ISAIM (2006)"},{"key":"31_CR16","doi-asserted-by":"crossref","unstructured":"Moore, A.W., Atkeson, C.G.: Prioritized sweeping: reinforcement learning with less data and less time. Mach. Learn., 103\u2013130 (1993)","DOI":"10.1007\/BF00993104"},{"key":"31_CR17","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming. Wiley, Hoboken (2014)"},{"key":"31_CR18","unstructured":"Russell, S., Norvig, P.: Artifical intelligence: a modern approach. \nhttp:\/\/aima.cs.berkeley.edu\/\n\n (2002). Accessed 22 Oct 2004"},{"key":"31_CR19","doi-asserted-by":"crossref","unstructured":"Sigaud, O., Buffet, O.: Markov Decision Processes in Artificial Intelligence. Wiley, Hoboken (2013)","DOI":"10.1002\/9781118557426"},{"key":"31_CR20","doi-asserted-by":"crossref","unstructured":"Sutton, R.S.: Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In: Proceedings of the Seventh International Conference on Machine Learning, pp. 216\u2013224 (1990)","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"31_CR21","doi-asserted-by":"crossref","unstructured":"van Otterlo, M., Wiering, M.: Reinforcement learning and Markov decision processes. Reinforcement Learning, pp. 3\u201342. Springer, Berlin (2012)","DOI":"10.1007\/978-3-642-27645-3_1"},{"issue":"11","key":"31_CR22","doi-asserted-by":"publisher","first-page":"1073","DOI":"10.1057\/jors.1993.181","volume":"44","author":"DJ White","year":"1993","unstructured":"White, D.J.: A survey of applications of Markov decision processes. J. Oper. Res. Soc. 44(11), 1073\u20131096 (1993)","journal-title":"J. Oper. Res. Soc."},{"key":"31_CR23","first-page":"851","volume":"6","author":"D Wingate","year":"2005","unstructured":"Wingate, D., Seppi, K.D.: Prioritization methods for accelerating MDP solvers. J. Mach. Learn. Res. 6, 851\u2013881 (2005)","journal-title":"J. Mach. Learn. Res."}],"container-title":["Springer Proceedings in Advanced Robotics","Robotics Research"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-28619-4_31","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,4,9]],"date-time":"2020-04-09T21:16:54Z","timestamp":1586467014000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-28619-4_31"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,28]]},"ISBN":["9783030286187","9783030286194"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-28619-4_31","relation":{},"ISSN":["2511-1256","2511-1264"],"issn-type":[{"type":"print","value":"2511-1256"},{"type":"electronic","value":"2511-1264"}],"subject":[],"published":{"date-parts":[[2019,11,28]]},"assertion":[{"value":"28 November 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}