{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T13:21:09Z","timestamp":1773235269952,"version":"3.50.1"},"publisher-location":"Berlin, Heidelberg","reference-count":15,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540678397","type":"print"},{"value":"9783540449140","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2000]]},"DOI":"10.1007\/3-540-44914-0_2","type":"book-chapter","created":{"date-parts":[[2007,5,22]],"date-time":"2007-05-22T21:26:14Z","timestamp":1179869174000},"page":"26-44","source":"Crossref","is-referenced-by-count":44,"title":["An Overview of MAXQ Hierarchical Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Thomas G.","family":"Dietterich","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2000,8,11]]},"reference":[{"key":"2_CR1","volume-title":"Neuro-Dynamic Programming","author":"D. P. Bertsekas","year":"1996","unstructured":"Bertsekas, D. P., & Tsitsiklis, J. N. (1996). Neuro-Dynamic Programming. Athena Scientific, Belmont, MA."},{"key":"2_CR2","first-page":"1017","volume-title":"Advances in Neural Information Processing Systems","author":"R. H. Crites","year":"1995","unstructured":"Crites, R. H., & Barto, A. G. (1995). Improving elevator performance using reinforcement learning. In Advances in Neural Information Processing Systems, Vol. 8, pp. 1017\u20131023 San Francisco, CA. Morgan Kaufmann."},{"key":"2_CR3","first-page":"271","volume-title":"Advances in Neural Information Processing Systems","author":"P. Dayan","year":"1993","unstructured":"Dayan, P., & Hinton, G. (1993). Feudal reinforcement learning. In Advances in Neural Information Processing Systems, 5, pp. 271\u2013278. Morgan Kaufmann, San Francisco, CA."},{"key":"2_CR4","volume-title":"Tech. rep. CS-95-10","author":"T. Dean","year":"1995","unstructured":"Dean, T., & Lin, S.-H. (1995). Decomposition techniques for planning in stochastic domains. Tech. rep. CS-95-10, Department of Computer Science, Brown University, Providence, Rhode Island."},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Dietterich, T. G. (2000). Hierarchical reinforcement learning with the MAXQ value function decomposition. Journal of Artificial Intelligence Research. To appear.","DOI":"10.1613\/jair.639"},{"key":"2_CR6","first-page":"167","volume-title":"Proceedings of the Tenth International Conference on Machine Learning","author":"L. P. Kaelbling","year":"1993","unstructured":"Kaelbling, L. P. (1993). Hierarchical reinforcement learning: Preliminary results. In Proceedings of the Tenth International Conference on Machine Learning, pp. 167\u2013173 San Francisco, CA. Morgan Kaufmann."},{"key":"2_CR7","first-page":"103","volume":"13","author":"A. W. Moore","year":"1993","unstructured":"Moore, A. W., & Atkeson, C. G. (1993). Prioritized sweeping: Reinforcement learning with less data and less time. Machine Learning, 13, 103.","journal-title":"Machine Learning"},{"key":"2_CR8","series-title":"Ph.D. thesis","volume-title":"Hierarchical control and learning for Markov decision processes","author":"R. Parr","year":"1998","unstructured":"Parr, R. (1998). Hierarchical control and learning for Markov decision processes. Ph.D. thesis, University of California, Berkeley, California."},{"key":"2_CR9","first-page":"1043","volume-title":"Advances in Neural Information Processing Systems","author":"R. Parr","year":"1998","unstructured":"Parr, R., & Russell, S. (1998). Reinforcement learning with hierarchies of machines. In Advances in Neural Information Processing Systems, Vol. 10, pp. 1043\u20131049 Cambridge, MA. MIT Press."},{"key":"2_CR10","volume-title":"Tech. rep.","author":"S. Singh","year":"1998","unstructured":"Singh, S., Jaakkola, T., Littman, M. L., & Szepesv\u00e1ri, C. (1998). Convergence results for single-step on-policy reinforcement-learning algorithms. Tech. rep., University of Colorado, Department of Computer Science, Boulder, CO. To appear in Machine Learning."},{"key":"2_CR11","first-page":"323","volume":"8","author":"S. P. Singh","year":"1992","unstructured":"Singh, S. P. (1992). Transfer of learning by composing solutions of elemental sequential tasks. Machine Learning, 8, 323.","journal-title":"Machine Learning"},{"key":"2_CR12","volume-title":"Introduction to Reinforcement Learning","author":"R. Sutton","year":"1998","unstructured":"Sutton, R., & Barto, A. G. (1998). Introduction to Reinforcement Learning. MIT Press, Cambridge, MA."},{"key":"2_CR13","volume-title":"Tech. rep.","author":"R. S. Sutton","year":"1998","unstructured":"Sutton, R. S., Precup, D., & Singh, S. (1998). Between MDPs and Semi-MDPs: Learning, planning, and representing knowledge at multiple temporal scales. Tech. rep., University of Massachusetts, Department of Computer and Information Sciences, Amherst, MA. To appear in Artificial Intelligence."},{"issue":"3","key":"2_CR14","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/203330.203343","volume":"28","author":"G. Tesauro","year":"1995","unstructured":"Tesauro, G. (1995). Temporal difference learning and TD-Gammon. Communications of the ACM, 28(3), 58\u201368.","journal-title":"Communications of the ACM"},{"key":"2_CR15","first-page":"1114","volume-title":"1995 International Joint Conference on Artificial Intelligence","author":"W. Zhang","year":"1995","unstructured":"Zhang, W., & Dietterich, T. G. (1995). A reinforcement learning approach to job-shop scheduling. In 1995 International Joint Conference on Artificial Intelligence, pp. 1114\u20131120. Morgan Kaufmann, San Francisco, CA."}],"container-title":["Lecture Notes in Computer Science","Abstraction, Reformulation, and Approximation"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/3-540-44914-0_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,2,16]],"date-time":"2019-02-16T22:21:29Z","timestamp":1550355689000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/3-540-44914-0_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2000]]},"ISBN":["9783540678397","9783540449140"],"references-count":15,"URL":"https:\/\/doi.org\/10.1007\/3-540-44914-0_2","relation":{},"ISSN":["0302-9743"],"issn-type":[{"value":"0302-9743","type":"print"}],"subject":[],"published":{"date-parts":[[2000]]}}}