{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T17:30:17Z","timestamp":1743096617671,"version":"3.40.3"},"publisher-location":"Boston, MA","reference-count":21,"publisher":"Springer US","isbn-type":[{"type":"print","value":"9780387307688"},{"type":"electronic","value":"9780387301648"}],"license":[{"start":{"date-parts":[[2011,1,1]],"date-time":"2011-01-01T00:00:00Z","timestamp":1293840000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2011,1,1]],"date-time":"2011-01-01T00:00:00Z","timestamp":1293840000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011]]},"DOI":"10.1007\/978-0-387-30164-8_556","type":"book-chapter","created":{"date-parts":[[2010,12,29]],"date-time":"2010-12-29T17:30:36Z","timestamp":1293643836000},"page":"690-693","source":"Crossref","is-referenced-by-count":4,"title":["Model-Based Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Soumya","family":"Ray","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Prasad","family":"Tadepalli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"556_CR1_556","volume-title":"Advances in neural information processing systems (Vol.\u00a019, pp.\u00a01\u20138)","author":"P Abbeel","year":"2007","unstructured":"Abbeel, P., Coates, A., Quigley, M., & Ng, A. Y. (2007). An application of reinforcement learning to aerobatic helicopter flight. In Advances in neural information processing systems (Vol.\u00a019, pp.\u00a01\u20138). Cambridge, MA: MIT Press."},{"key":"556_CR2_556","volume-title":"Proceedings of the 23rd international conference on machine learning (pp. 1\u20138)","author":"P Abbeel","year":"2006","unstructured":"Abbeel, P., Quigley, M., & Ng, A. Y. (2006). Using inaccurate models in reinforcement learning. In Proceedings of the 23rd international conference on machine learning (pp. 1\u20138). ACM Press, New York, USA."},{"key":"556_CR3_556","doi-asserted-by":"crossref","unstructured":"Atkeson, C. G., & Santamaria, J. C. (1997). A comparison of direct and model-based reinforcement learning. In Proceedings of the international conference on robotics and automation (pp. 20\u201325). IEEE Press.","DOI":"10.1109\/ROBOT.1997.606886"},{"key":"556_CR4_556","volume-title":"Proceedings of the fourteenth international conference on machine learning (Vol.\u00a04, pp. 12\u201320)","author":"CG Atkeson","year":"1997","unstructured":"Atkeson, C. G., & Schaal, S. (1997). Robot learning from demonstration. In Proceedings of the fourteenth international conference on machine learning (Vol.\u00a04, pp. 12\u201320). San Francisco:\u00a0Morgan Kaufmann."},{"issue":"1","key":"556_CR5_556","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"AG Barto","year":"1995","unstructured":"Barto, A. G., Bradtke, S. J., & Singh, S. P. (1995). Learning to act using real-time dynamic programming. Artificial Intelligence, 72(1), 81\u2013138.","journal-title":"Artificial Intelligence"},{"key":"556_CR6_556","unstructured":"Baxter, J., Tridgell, A., & Weaver, L. (1998). TDLeaf(\u03bb): Combining temporal difference learning with game-tree search. In Proceedings of the ninth Australian conference on neural networks (ACNN\u201998) (pp. 168\u2013172)."},{"key":"556_CR7_556","first-page":"213","volume":"2","author":"RI Brafman","year":"2002","unstructured":"Brafman, R. I., & Tennenholtz, M. (2002). R-MAX \u2013 a general polynomial time algorithm for near-optimal reinforcement learning. Journal of Machine Learning Research, 2, 213\u2013231.","journal-title":"Journal of Machine Learning Research"},{"key":"556_CR8_556","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling, L. P., Littman, M. L., & Moore, A. P. (1996). Reinforcement learning: A survey. Journal of Artificial Intelligence Research, 4, 237\u2013285.","journal-title":"Journal of Artificial Intelligence Research"},{"issue":"2\/3","key":"556_CR9_556","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1023\/A:1017984413808","volume":"49","author":"M Kearns","year":"2002","unstructured":"Kearns, M., & Singh, S. (2002). Near-optimal reinforcement learning in polynomial time. Machine Learning, 49(2\/3), 209\u2013232.","journal-title":"Machine Learning"},{"key":"556_CR10_556","first-page":"103","volume":"13","author":"AW Moore","year":"1993","unstructured":"Moore, A. W., & Atkeson, C. G. (1993). Prioritized sweeping: Reinforcement learning with less data and less real time. Machine Learning, 13, 103\u2013130.","journal-title":"Machine Learning"},{"issue":"4","key":"556_CR11_556","doi-asserted-by":"crossref","first-page":"437","DOI":"10.1177\/105971239300100403","volume":"1","author":"J Peng","year":"1993","unstructured":"Peng, J., & Williams, R. J. (1993). Efficient learning and planning within the dyna framework. Adaptive Behavior, 1(4), 437\u2013454.","journal-title":"Adaptive Behavior"},{"key":"556_CR12_556","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","volume-title":"Markov decision processes: Discrete dynamic stochastic programming","author":"ML Puterman","year":"1994","unstructured":"Puterman, M. L. (1994). Markov decision processes: Discrete dynamic stochastic programming. New York: Wiley."},{"issue":"1","key":"556_CR13_556","doi-asserted-by":"crossref","first-page":"57","DOI":"10.1109\/37.257895","volume":"14","author":"S Schaal","year":"1994","unstructured":"Schaal, S., & Atkeson, C. G. (1994). Robot juggling: Implementation of memory-based learning. IEEE Control Systems Magazine, 14(1), 57\u201371.","journal-title":"IEEE Control Systems Magazine"},{"key":"556_CR14_556","unstructured":"Singh, S., Kearns, M., Litman, D., & Walker, M. (1999) Reinforcement learning for spoken dialogue systems. In Advances in neural information processing systems (Vol.\u00a011, pp. 956\u2013962). MIT Press."},{"key":"556_CR15_556","first-page":"216","volume-title":"Proceedings of the seventh international conference on machine learning","author":"RS Sutton","year":"1990","unstructured":"Sutton, R. S. (1990). Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In Proceedings of the seventh international conference on machine learning (pp. 216\u2013224). San Francisco: Morgan Kaufmann."},{"key":"556_CR16_556","volume-title":"Reinforcement learning: An introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning: An introduction. Cambridge, MA: MIT Press."},{"key":"556_CR17_556","doi-asserted-by":"crossref","first-page":"177","DOI":"10.1016\/S0004-3702(98)00002-2","volume":"100","author":"P Tadepalli","year":"1998","unstructured":"Tadepalli, P., & Ok, D. (1998). Model-based average-reward reinforcement learning. Artificial Intelligence, 100, 177\u2013224.","journal-title":"Artificial Intelligence"},{"issue":"3","key":"556_CR18_556","doi-asserted-by":"crossref","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G Tesauro","year":"1995","unstructured":"Tesauro, G. (1995). Temporal difference learning and TD-Gammon. Communications of the ACM, 38(3), 58\u201368.","journal-title":"Communications of the ACM"},{"key":"556_CR19_556","unstructured":"Wang, X., & Dietterich, T. G. (2003). Model-based policy gradient reinforcement learning. In Proceedings of the 20th international conference on machine learning (pp. 776\u2013783). AAAI Press."},{"key":"556_CR20_556","doi-asserted-by":"crossref","first-page":"1015","DOI":"10.1145\/1273496.1273624","volume-title":"Proceedings of the 24th international conference on machine learning","author":"A Wilson","year":"2007","unstructured":"Wilson, A., Fern, A., Ray, S., & Tadepalli, P. (2007). Multi-task reinforcement learning: A hierarchical Bayesian approach. In Proceedings of the 24th international conference on machine learning (pp. 1015\u20131022). Madison, WI: Omnipress."},{"key":"556_CR21_556","unstructured":"Zhang, W., & Dietterich, T. G. (1995). A reinforcement learning approach to job-shop scheduling. In Proceedings of the international joint conference on artificial intelligence (pp. 1114\u20131120). Morgan Kaufman."}],"container-title":["Encyclopedia of Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-0-387-30164-8_556","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,22]],"date-time":"2023-12-22T02:41:07Z","timestamp":1703212867000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-0-387-30164-8_556"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011]]},"ISBN":["9780387307688","9780387301648"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-0-387-30164-8_556","relation":{},"subject":[],"published":{"date-parts":[[2011]]}}}