{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T17:07:01Z","timestamp":1742922421498,"version":"3.40.3"},"publisher-location":"Boston, MA","reference-count":21,"publisher":"Springer US","isbn-type":[{"type":"print","value":"9781489976857"},{"type":"electronic","value":"9781489976871"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-1-4899-7687-1_561","type":"book-chapter","created":{"date-parts":[[2017,4,13]],"date-time":"2017-04-13T12:32:16Z","timestamp":1492086736000},"page":"852-855","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Model-Based Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Soumya","family":"Ray","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Prasad","family":"Tadepalli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,4,14]]},"reference":[{"key":"561_CR13251","first-page":"1","volume-title":"Advances in neural information processing systems","author":"P Abbeel","year":"2007","unstructured":"Abbeel P, Coates A, Quigley M, Ng AY (2007) An application of reinforcement learning to aerobatic helicopter flight. In: Advances in neural information processing systems, vol\u00a019. MIT, Cambridge, pp\u00a01\u20138"},{"key":"561_CR13252","first-page":"1","volume-title":"Proceedings of the 23rd international conference on machine learning","author":"P Abbeel","year":"2006","unstructured":"Abbeel P, Quigley M, Ng AY (2006) Using inaccurate models in reinforcement learning. In: Proceedings of the 23rd international conference on machine learning, Pittsburgh. ACM, New York, pp\u00a01\u20138"},{"key":"561_CR13253","first-page":"20","volume-title":"Proceedings of the international conference on robotics and automation","author":"CG Atkeson","year":"1997","unstructured":"Atkeson CG, Santamaria JC (1997) A comparison of direct and model-based reinforcement learning. In: Proceedings of the international conference on robotics and automation, Albuquerque. IEEE, pp\u00a020\u201325"},{"key":"561_CR13254","first-page":"12","volume-title":"Proceedings of the fourteenth international conference on machine learning, Nashville","author":"CG Atkeson","year":"1997","unstructured":"Atkeson CG, Schaal S (1997) Robot learning from demonstration. In: Proceedings of the fourteenth international conference on machine learning, Nashville, vol\u00a04. Morgan Kaufmann, San Francisco, pp\u00a012\u201320"},{"issue":"1","key":"561_CR13255","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"AG Barto","year":"1995","unstructured":"Barto AG, Bradtke SJ, Singh SP (1995) Learning to act using real-time dynamic programming. Artif Intell 72(1):81\u2013138","journal-title":"Artif Intell"},{"key":"561_CR13256","unstructured":"Baxter J, Tridgell A, Weaver L (1998) TDLeaf(\u03bb): combining temporal difference learning with game-tree search. In: Proceedings of the ninth Australian conference on neural networks (ACNN\u201998), Brisbane, pp\u00a0168\u2013172"},{"key":"561_CR13257","first-page":"213","volume":"2","author":"RI Brafman","year":"2002","unstructured":"Brafman RI, Tennenholtz M (2002) R-MAX \u2013 a general polynomial time algorithm for near-optimal reinforcement learning. J Mach Learn Res 2:213\u2013231","journal-title":"J Mach Learn Res"},{"key":"561_CR13258","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling LP, Littman ML, Moore AP (1996) Reinforcement learning: a survey. J Artif Intell Res 4:237\u2013285","journal-title":"J Artif Intell Res"},{"issue":"2\/3","key":"561_CR13259","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1023\/A:1017984413808","volume":"49","author":"M Kearns","year":"2002","unstructured":"Kearns M, Singh S (2002) Near-optimal reinforcement learning in polynomial time. Mach Learn 49(2\/3):209\u2013232","journal-title":"Mach Learn"},{"key":"561_CR13260","first-page":"103","volume":"13","author":"AW Moore","year":"1993","unstructured":"Moore AW, Atkeson CG (1993) Prioritized sweeping: reinforcement learning with less data and less real time. Mach Learn 13:103\u2013130","journal-title":"Mach Learn"},{"issue":"4","key":"561_CR13261","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1177\/105971239300100403","volume":"1","author":"J Peng","year":"1993","unstructured":"Peng J, Williams RJ (1993) Efficient learning and planning within the Dyna framework. Adapt Behav 1(4):437\u2013454","journal-title":"Adapt Behav"},{"key":"561_CR13262","doi-asserted-by":"publisher","DOI":"10.1002\/9780470316887","volume-title":"Markov decision processes: discrete dynamic stochastic programming","author":"ML Puterman","year":"1994","unstructured":"Puterman ML (1994) Markov decision processes: discrete dynamic stochastic programming. Wiley, New York"},{"issue":"1","key":"561_CR13263","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1109\/37.257895","volume":"14","author":"S Schaal","year":"1994","unstructured":"Schaal S, Atkeson CG (1994) Robot juggling: implementation of memory-based learning. IEEE Control Syst Mag 14(1):57\u201371","journal-title":"IEEE Control Syst Mag"},{"key":"561_CR13264","unstructured":"Singh S, Kearns M, Litman D, Walker M (1999) Reinforcement learning for spoken dialogue systems. In: Advances in neural information processing systems, Denver, vol\u00a011. MIT, pp\u00a0956\u2013962"},{"key":"561_CR13265","first-page":"216","volume-title":"Proceedings of the seventh international conference on machine learning","author":"RS Sutton","year":"1990","unstructured":"Sutton RS (1990) Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In: Proceedings of the seventh international conference on machine learning, Austin. Morgan Kaufmann, San Francisco, pp\u00a0216\u2013224"},{"key":"561_CR13266","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction. MIT, Cambridge"},{"key":"561_CR13267","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1016\/S0004-3702(98)00002-2","volume":"100","author":"P Tadepalli","year":"1998","unstructured":"Tadepalli P, Ok D (1998) Model-based average-reward reinforcement learning. Artif Intell 100:177\u2013224","journal-title":"Artif Intell"},{"issue":"3","key":"561_CR13268","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro G (1995) Temporal difference learning and TD-Gammon. Commun ACM 38(3):58\u201368","journal-title":"Commun ACM"},{"key":"561_CR13269","first-page":"776","volume-title":"Model-based policy gradient reinforcement learning","author":"X Wang","year":"2003","unstructured":"Wang X, Dietterich TG (2003) Model-based policy gradient reinforcement learning. In: Proceedings of the 20th international conference on machine learning, Washington, DC. AAAI, pp\u00a0776\u2013783"},{"key":"561_CR13270","doi-asserted-by":"crossref","first-page":"1015","DOI":"10.1145\/1273496.1273624","volume-title":"Proceedings of the 24th international conference on machine learning","author":"A Wilson","year":"2007","unstructured":"Wilson A, Fern A, Ray S, Tadepalli P (2007) Multi-task reinforcement learning: a hierarchical Bayesian approach. In: Proceedings of the 24th international conference on machine learning, Corvalis. Omnipress, Madison, pp\u00a01015\u20131022"},{"key":"561_CR13271","first-page":"1114","volume-title":"Proceedings of the international joint conference on artificial intelligence","author":"W Zhang","year":"1995","unstructured":"Zhang W, Dietterich TG (1995) A reinforcement learning approach to job-shop scheduling. In: Proceedings of the international joint conference on artificial intelligence, Montr\u00e9al. Morgan Kaufman, pp\u00a01114\u20131120"}],"container-title":["Encyclopedia of Machine Learning and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-1-4899-7687-1_561","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,27]],"date-time":"2022-07-27T13:31:54Z","timestamp":1658928714000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-1-4899-7687-1_561"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9781489976857","9781489976871"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-1-4899-7687-1_561","relation":{},"subject":[],"published":{"date-parts":[[2017]]},"assertion":[{"value":"14 April 2017","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}