{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T00:06:18Z","timestamp":1785197178510,"version":"3.55.0"},"reference-count":88,"publisher":"Springer Science and Business Media LLC","issue":"1-2","license":[{"start":{"date-parts":[[2003,1,1]],"date-time":"2003-01-01T00:00:00Z","timestamp":1041379200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2003,1,1]],"date-time":"2003-01-01T00:00:00Z","timestamp":1041379200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Discrete Event Dynamic Systems"],"published-print":{"date-parts":[[2003,1]]},"DOI":"10.1023\/a:1022140919877","type":"journal-article","created":{"date-parts":[[2003,3,21]],"date-time":"2003-03-21T19:29:05Z","timestamp":1048274945000},"page":"41-77","source":"Crossref","is-referenced-by-count":445,"title":["Recent Advances in Hierarchical Reinforcement Learning"],"prefix":"10.1007","volume":"13","author":[{"given":"Andrew G.","family":"Barto","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sridhar","family":"Mahadevan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"key":"5110821_CR1","first-page":"1019","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 2000 Conference","author":"D. Andre","year":"2001","unstructured":"Andre, D., and Russell, S. J. 2001. Programmable reinforcement learning agents. In Advances in Neural Information Processing Systems: Proceedings of the 2000 Conference. Cambridge, MA: MIT Press, pp. 1019\u20131025."},{"key":"5110821_CR2","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"A. G. Barto","year":"1995","unstructured":"Barto, A. G., Bradtke, S. J., and Singh, S. P. 1995. Learning to act using real-time dynamic programming. Artificial Intelligence 72: 81\u2013138.","journal-title":"Artificial Intelligence"},{"key":"5110821_CR3","unstructured":"Bernstein, D., Zilberstein, S., and Immerman, N. 2000. The complexity of decentralized control of markov decision processes. In 16th Conference on Uncertainty in Artificial Intelligence."},{"key":"5110821_CR4","volume-title":"Dynamic Programming: Deterministic and Stochastic Models","author":"D. P. Bertsekas","year":"1987","unstructured":"Bertsekas, D. P. 1987. Dynamic Programming: Deterministic and Stochastic Models. Prentice-Hall, Englewood Cliffs, NJ."},{"key":"5110821_CR5","volume-title":"Neuro-Dynamic Programming","author":"D. P. Bertsekas","year":"1996","unstructured":"Bertsekas, D. P., and Tsitsiklis, J. N. 1996. Neuro-Dynamic Programming. Belmont, MA: Athena Scientific."},{"key":"5110821_CR6","first-page":"33","volume-title":"Proceedings of the Fourteenth Conference on Uncertainty in AI","author":"X. Boyen","year":"1998","unstructured":"Boyen, X., and Koller, D. 1998. Tractable inference for complex stochastic processes. In G. F. Cooper and S. Moral, editors, Proceedings of the Fourteenth Conference on Uncertainty in AI. San Francisco, CA, Morgan Kaufmann, pp. 33\u201342."},{"key":"5110821_CR7","first-page":"393","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1994 Conference","author":"S. J. Bradtke","year":"1995","unstructured":"Bradtke, S. J., and Duff, M. O. 1995. Reinforcement learning methods for continuous-time Markov decision problems. In G. Tesauro, D. S. Touretzky, and T. Leen, editors, Advances in Neural Information Processing Systems: Proceedings of the 1994 Conference, Cambridge, MA: MIT Press, pp. 393\u2013400."},{"key":"5110821_CR8","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1109\/9.654885","volume":"43","author":"M. S. Branicky","year":"1998","unstructured":"Branicky, M. S., Borkar, V. S., and Mitter, S. K. 1998. A unified framework for hybrid control: Model and optimal control theory. IEEE Transactions on Automatic Control 43: 31\u201345.","journal-title":"IEEE Transactions on Automatic Control"},{"key":"5110821_CR9","series-title":"Technical Report","volume-title":"Achieving Artificial Intelligence through building robots","author":"R. A. Brooks","year":"1986","unstructured":"Brooks, R. A. 1986. Achieving Artificial Intelligence through building robots. Technical Report A.I. Memo 899, Massachusetts Institute of Technology Artificial Intelligence Laboratory, Cambridge, MA."},{"key":"5110821_CR10","volume-title":"Large-Scale Dynamic Optimization Using Teams of Reinforcement Learning Agents","author":"R. H. Crites","year":"1996","unstructured":"Crites, R. H. 1996. Large-Scale Dynamic Optimization Using Teams of Reinforcement Learning Agents. Ph.D. thesis, Amberst, MA: University of Massachusetts."},{"key":"5110821_CR11","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1007518724497","volume":"33","author":"R. H. Crites","year":"1998","unstructured":"Crites, R. H., and Barto, A. G. 1998. Elevator group control using multiple reinforcement learning agents. Machine Learning 33: 235\u2013262.","journal-title":"Machine Learning"},{"key":"5110821_CR12","doi-asserted-by":"crossref","first-page":"560","DOI":"10.1287\/mnsc.45.4.560","volume":"45","author":"T. K. Das","year":"1999","unstructured":"Das, T. K., Gosavi, A., Mahadevan, S., and Marchalleck, N. 1999. Solving semi-Markov decision problems using average reward reinforcement learning. Management Science 45: 560\u2013574.","journal-title":"Management Science"},{"key":"5110821_CR13","doi-asserted-by":"crossref","first-page":"142","DOI":"10.1111\/j.1467-8640.1989.tb00324.x","volume":"5","author":"T. L. Dean","year":"1989","unstructured":"Dean, T. L., and Kanazawa, K. 1989. A model for reasoning about persistence and causation. Computational Intelligence 5: 142\u2013150.","journal-title":"Computational Intelligence"},{"key":"5110821_CR14","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"T. G. Dietterich","year":"2000","unstructured":"Dietterich, T. G. 2000. Hierarchical reinforcement learning with the maxq value function decomposition. Journal of Artificial Intelligence Research 13: 227\u2013303.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"5110821_CR15","volume-title":"From Animals to Animats 4: The Fourth Conference on Simulation of Adaptive Behavior","author":"B. Digney","year":"1996","unstructured":"Digney, B. 1996. Emergent hierarchical control structures: Learning reactive\/hierarchical relationships in reinforcement environments. In P. Meas and M. Mataric, editors, From Animals to Animats 4: The Fourth Conference on Simulation of Adaptive Behavior. Cambridge, MA: MIT Press."},{"key":"5110821_CR16","volume-title":"From Animals to Animals 5: The Fifth Conference on Simulation of Adaptive Behavior","author":"B. Digney","year":"1998","unstructured":"Digney, B. 1998. Learning hierarchical control structure from multiple tasks and changing environments. In From Animals to Animals 5: The Fifth Conference on Simulation of Adaptive Behavior. Cambridge, MA: MIT Press."},{"key":"5110821_CR17","unstructured":"Driessens, K., and Dzeroski, S. 2002. Integrating experimentation and guidance in relational reinforcement learning. In Machine Learning: Proceedings of the Nineteenth International Conference on Machine Learning."},{"key":"5110821_CR18","doi-asserted-by":"crossref","first-page":"251","DOI":"10.1016\/0004-3702(72)90051-3","volume":"3","author":"R. E. Fikes","year":"1972","unstructured":"Fikes, R. E., Hart, P. E., and Nilsson, N. J. 1972. Learning and executing generalized robot plans. Artificial Intelligence 3: 251\u2013288.","journal-title":"Artificial Intelligence"},{"key":"5110821_CR19","doi-asserted-by":"crossref","unstructured":"Finc, S., Singer, Y., and Tishby, N. 1998. The hierarchical hidden Markov model: analysis and applications. Machine Learning 32(1): July.","DOI":"10.1023\/A:1007469218079"},{"key":"5110821_CR20","doi-asserted-by":"crossref","first-page":"298","DOI":"10.1109\/TAC.1978.1101707","volume":"23","author":"J.-P. Forestier","year":"1978","unstructured":"Forestier, J.-P., and Varaiya, P. 1978. Multilayer control of large Markov chains. IEEE Transactions on Automatic Control AC-23: 298\u2013304.","journal-title":"IEEE Transactions on Automatic Control AC-"},{"key":"5110821_CR21","unstructured":"Ghavamzadeh, M., and Mahadevan, S. 2001. Continuous-time hierarchical reinforcement learning. In Proceedings of the Eighteenth International Conference on Machine Learning."},{"key":"5110821_CR22","unstructured":"Grudic, G. Z., and Ungar, L. H. 2000. Localizing search in reinforcement learning. In Proceedings of the 18th National Conference on Artificial Intelligence (AAAI-00), pp. 590\u2013595."},{"key":"5110821_CR23","doi-asserted-by":"crossref","first-page":"231","DOI":"10.1016\/0167-6423(87)90035-9","volume":"8","author":"D. Harel","year":"1987","unstructured":"Harel, D. 1987. Statecharts: A visual formalixm for complex systems. Science of Computer Programming 8: 231\u2013274.","journal-title":"Science of Computer Programming"},{"key":"5110821_CR24","unstructured":"Hengst, B. 2002. Discovering hierarchy in reinforcement learning with hexq. In Machine Learning: Proceedings of the Nineteenth International Conference on Machine Learning."},{"key":"5110821_CR25","unstructured":"Hernandez, N., and Mahadevan, S. 2001. Hierarchical memory-based reinforcement learning. Proceedings of Neural Information Processing Systems."},{"key":"5110821_CR26","volume-title":"Dynamic Probabilistic Systems: Semi-Markov and Decision Processes","author":"R. A. Howard","year":"1971","unstructured":"Howard, R. A. 1971. Dynamic Probabilistic Systems: Semi-Markov and Decision Processes. New York: Wiley."},{"key":"5110821_CR27","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1016\/S0921-8890(97)00044-4","volume":"22","author":"M. Huber","year":"1997","unstructured":"Huber, M., and Grupen, R. A. 1997. A feedback control structure for on-line learning tasks. Robotics and Autonomous Systems 22: 303\u2013315.","journal-title":"Robotics and Autonomous Systems"},{"key":"5110821_CR28","doi-asserted-by":"crossref","first-page":"285","DOI":"10.1023\/A:1022693717366","volume":"3","author":"G. A. Iba","year":"1989","unstructured":"Iba, G. A. 1989. A heuristic approach to the discovery of macro-operators. Machine Learning 3: 285\u2013317.","journal-title":"Machine Learning"},{"key":"5110821_CR29","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1162\/neco.1994.6.6.1185","volume":"6","author":"T. Jaakkola","year":"1994","unstructured":"Jaakkola, T., Jordan, M. I., and Singh, S. P. 1994. On the convergence of stochastic iterative dynamic programming algorithms. Neural Computation 6: 1185\u20131201.","journal-title":"Neural Computation"},{"key":"5110821_CR30","first-page":"1054","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 2000 Conference","author":"A. Jonsson","year":"2001","unstructured":"Jonsson, A., and Barto, A. G. 2001. Automated state abstraction for options using the U-tree algorithm. In Advances in Neural Information Processing Systems: Proceedings of the 2000 Conference, Cambridge, MA: MIT Press, pp. 1054\u20131060."},{"key":"5110821_CR31","doi-asserted-by":"crossref","unstructured":"Kaelbling, L., Littman, M., and Cassandra A. 1998. Planning and acting in partially observable stochastic domains. Artificial Intelligence 101.","DOI":"10.1016\/S0004-3702(98)00023-X"},{"key":"5110821_CR32","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"L. P. Kaelbling","year":"1996","unstructured":"Kaelbling, L. P., Littman, M. L., and Moore, A. W. 1996. Reinforcement learning: A survey. Journal of Artificial Intelligence Research 4: 237\u2013285.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"5110821_CR33","unstructured":"Klopf, A. H. 1974. Brain function and adaptive systems-\u00d0A heterostatic theory. Technical Report AFCRL\u201372\u20130164, Air Force Cambridge Research Laboratories, Bedford, MA, 1972. A summary appears in Proceedings of the International Conference on Systems, Man, and Cybernetics, IEEE Systems, Man, and Cybernetics Society, Dallas, TX"},{"key":"5110821_CR34","volume-title":"The Hedonistic Neuron: A Theory of Memory, Learning, and Intelligence","author":"A. H. Klopf","year":"1982","unstructured":"Klopf, A. H. 1982. The Hedonistic Neuron: A Theory of Memory, Learning, and Intelligence. Washington, D.C.: Hemisphere."},{"key":"5110821_CR35","volume-title":"Al-based Mobile Robots: Case-studies of Successful Robot Systems","author":"S. Koening","year":"1997","unstructured":"Koening, S., and Simmons, R. 1997. Xavier: A robot navigation architecture based on partially observable Markov decision process models. In D. Kortenkamp, P. Bonasso, and R. Murphy, editors, Al-based Mobile Robots: Case-studies of Successful Robot Systems. Cambridge, MA: MIT Press."},{"key":"5110821_CR36","volume-title":"Singular Perturbation Methods in Control: Analysis and Design","author":"P. V. Kokotovic","year":"1986","unstructured":"Kokotovic, P. V., Khalil, H. K., and O'Reilly, J. 1986. Singular Perturbation Methods in Control: Analysis and Design. London: Academic Press."},{"key":"5110821_CR37","volume-title":"Learning to Solve Problems by Searching for Macro-Operators","author":"R. E. Korf","year":"1985","unstructured":"Korf, R. E. 1985. Learning to Solve Problems by Searching for Macro-Operators. Boston, MA: Pitman."},{"key":"5110821_CR38","doi-asserted-by":"crossref","unstructured":"Littman, M. 1994. Markov games as a framework for multi-agent reinforcement learning. In Proceedings of the Eleventh International Conference on Machine Learning, pp. 157\u2013163.","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"5110821_CR39","doi-asserted-by":"crossref","first-page":"159","DOI":"10.1023\/A:1018064306595","volume":"22","author":"S. Mahadevan","year":"1996","unstructured":"Mahadevan, S. 1996. Average reward reinforcement learning: Foundations, algorithms, and empirical results. Machine Learning 22: 159\u2013196.","journal-title":"Machine Learning"},{"key":"5110821_CR40","unstructured":"Mahadevan, S., Marchalleck, N., Das, T., and Gosavi, A. 1997. Self-improving factory simulation using continuous-time average-reward reinforcement learning. In Machine Learning: Proceedings of the Fourteenth International Conference."},{"key":"5110821_CR41","doi-asserted-by":"crossref","unstructured":"Makar, R., Mahadevan, S., and Ghavamzadeh, M. 2001. Hierarchical multi-agent reinforcement learning. In J. P. Mu\u00c8ller, E. Andre, S. Sen, and C. Frasson, editors, Proceedings of the Fifth International Conference on Autonomous Agents, pp. 246\u2013253.","DOI":"10.1145\/375735.376302"},{"key":"5110821_CR42","unstructured":"McCallum, A. K. 1996. Reinforcement Learning with Selective Perception and Hidden State. Ph.D. thesis, University of Rochester."},{"key":"5110821_CR43","doi-asserted-by":"crossref","unstructured":"McGovern, A. 2002. Autonomous Discovery of Temporal Abstractions from Interaction with An Environment. Ph.D. thesis, University of Massachusetts.","DOI":"10.1007\/3-540-45622-8_34"},{"key":"5110821_CR44","first-page":"361","volume-title":"Proceedings of the Eighteenth International Conference on Machine Learning","author":"A. McGovern","year":"2001","unstructured":"McGovern, A., and Barto, A. 2001. Automatic discovery of subgoals in reinforcement learning using diverse density. In C. Brodley and A. Danyluk, editors, Proceedings of the Eighteenth International Conference on Machine Learning, San Francisco, CA: Morgan Kaufmann, pp. 361\u2013368."},{"key":"5110821_CR45","unstructured":"Minsky, M. L. 1954. Theory of Neural-Analog Reinforcement Systems and its Application to the Brain-Model Problem. Ph.D. thesis, Princeton University."},{"key":"5110821_CR46","doi-asserted-by":"crossref","DOI":"10.1049\/PBCE034E","volume-title":"Singular Perturbation Methodology in Control Systems","author":"D. S. Naidu","year":"1988","unstructured":"Naidu, D. S. 1988. Singular Perturbation Methodology in Control Systems. London: Peter Peregrinus Ltd."},{"issue":"2","key":"5110821_CR47","first-page":"53","volume":"16","author":"I. Nourbakhsh","year":"1995","unstructured":"Nourbakhsh, I., Powers, R., and Birchfield, S. 1995. Dervish: An office-navigation robot. Al Magazine 16(2): 53\u201360.","journal-title":"Al Magazine"},{"key":"5110821_CR48","volume-title":"Hierarchical Control and Learning for Markov Decision Processes","author":"R. Parr","year":"1998","unstructured":"Parr, R. 1998. Hierarchical Control and Learning for Markov Decision Processes. Ph.D. Thesis, Berkeley CA: University of California."},{"key":"5110821_CR49","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1997 Conference","author":"R. Parr","year":"1998","unstructured":"Parr, R., and Russell, S. 1998. Reinforcement learning with hierarchies of machines. In Advances in Neural Information Processing Systems: Proceedings of the 1997 Conference. Cambridge, MA: MIT Press."},{"key":"5110821_CR50","unstructured":"Perkins, T. J., and Barto, A. G. Lyapunov design for safe reinforcement learning. Journal of Machine Learning Research. To appear."},{"key":"5110821_CR51","first-page":"409","volume-title":"Proceedings of the Eighteenth International Conference on Machine Learning","author":"T. J. Perkins","year":"2001","unstructured":"Perkins, T. J., and Barto, A. G. 2001. Lyapunov-constrained action sets for reinforcement learning. In C. Brodley and A. Danyluk, editors. Proceedings of the Eighteenth International Conference on Machine Learning. San Francisco, CA: Morgan Kaufmann, pp. 409\u2013416."},{"key":"5110821_CR52","volume-title":"Temporal Abstraction in Reinforcement Learning","author":"D. Precup","year":"2000","unstructured":"Precup, D. 2000. Temporal Abstraction in Reinforcement Learning. Ph.D. thesis, Amherst, MA: University of Massachusetts."},{"key":"5110821_CR53","first-page":"1050","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1997 Conference","author":"D. Precup","year":"1998","unstructured":"Precup, D., and Sutton, R. S. 1998. Multi-time models for temporally abstract planning. In: Advances in Neural Information Processing Systems: Proceedings of the 1997 Conference. Cambridge MA: MIT Press, pp. 1050\u20131056."},{"key":"5110821_CR54","doi-asserted-by":"crossref","unstructured":"Precup, D., Sutton, R. S., and Singh, S. 1998. Theoretical results on reinforcement learning with temporally abstract options. In Proceedings of the 10th European Conference on Machine Learning, ECML-98. Springer Verlag, pp. 382\u2013393.","DOI":"10.1007\/BFb0026709"},{"key":"5110821_CR55","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","volume-title":"Markov Decision Problems","author":"M. L. Puterman","year":"1994","unstructured":"Puterman, M. L. 1994. Markov Decision Problems. New York: Wiley."},{"key":"5110821_CR56","unstructured":"Rohanimanesh, K., and Mahadevan, S. Structured approximation of stochastic temporally extended actions. In preparation."},{"key":"5110821_CR57","unstructured":"Rohanimanesh, K., and Mahadevan, S.2001. Decision-theoretic planning with concurrent temporally extended actions. In Proceedings of the Seventeenth Conference on Uncertainty in Artificial Intelligence."},{"key":"5110821_CR58","volume-title":"Introduction to Stochastic Dynamic Programming","author":"S. Ross","year":"1983","unstructured":"Ross, S. 1983. Introduction to Stochastic Dynamic Programming. New York: Academic Press."},{"key":"5110821_CR59","unstructured":"Rummery, G. A., and Niranjan, M. 1994. On-line q-learning using connectionist systems. Technical Report CUED\/F-INFENG\/TR 166, Cambridge University Engineering Department."},{"key":"5110821_CR60","first-page":"211","volume-title":"IBM Journal on Research and Development","author":"A. L. Samuel","year":"1959","unstructured":"Samuel, A. L. 1959. Some studies in machine learning using the game of checkers. IBM Journal on Research and Development 3: 211\u2013229. Reprinted in E. A. Feigenbaum and J. Feldman, editors, Computers and Thought, New York: McGraw-Hill, pp. 71\u2013105."},{"key":"5110821_CR61","doi-asserted-by":"crossref","first-page":"601","DOI":"10.1147\/rd.116.0601","volume":"11","author":"A. L. Samuel","year":"1967","unstructured":"Samuel, A. L. 1967. Some studies in machine learning using the game of checkers. II\u2014Recent progress. IBM Journal on Research and Development 11: 601\u2013617.","journal-title":"IBM Journal on Research and Development"},{"key":"5110821_CR62","doi-asserted-by":"crossref","unstructured":"Schwartz, A. 1993. A reinforcement learning method for maximizing undiscounted rewards. In Proceedings of the Tenth International Conference on Machine Learning. Morgan Kaufmann, pp. 298\u2013305.","DOI":"10.1016\/B978-1-55860-307-3.50045-9"},{"key":"5110821_CR63","first-page":"920","volume":"2","author":"H. Shatkay","year":"1997","unstructured":"Shatkay, H., and Kaelbling, L. P. 1997. Learning topological maps with weak local odometric information. In IJCAI 2), pp. 920\u2013929.","journal-title":"IJCAI"},{"key":"5110821_CR64","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1996 Conference","author":"S. Singh","year":"1997","unstructured":"Singh, S., and Bertsekas, D. 1997. Reinforcement learning for dynamic channel allocation in cellular telephone systems. In Advances in Neural Information Processing Systems: Proceedings of the 1996 Conference. Cambridge, MA: MIT Press."},{"key":"5110821_CR65","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1023\/A:1007678930559","volume":"38","author":"S. Singh","year":"2000","unstructured":"Singh, S., Jaakkola, T., Littman, M. L., and Szepesva\u00c2ri. C. 2000. Convergence results for single-step on-policy reinforcement-learning algorithms. Machine Learning 38: 287\u2013308.","journal-title":"Machine Learning"},{"key":"5110821_CR66","first-page":"202","volume-title":"Proceedings of the Tenth National Conference on Artificial Intelligence","author":"S. P. Singh","year":"1992","unstructured":"Singh, S. P. 1992. Reinforcement learning with a hierarchy of abstract models. In Proceedings of the Tenth National Conference on Artificial Intelligence. Menlo Park, CA: AAAI Press\/MIT Press, pp. 202\u2013207."},{"key":"5110821_CR67","first-page":"406","volume-title":"Proceedings of the Ninth International Machine Learning Conference","author":"S. P. Singh","year":"1992","unstructured":"Singh, S. P. 1992. Scaling reinforcement learning algorithms by learning variable temporal resolution models. In Proceedings of the Ninth International Machine Learning Conference. San Mateo, CA: Morgan Kaufmann, pp. 406\u2013415."},{"key":"5110821_CR68","first-page":"537","volume-title":"Proceedings of the Eighteenth International Conference on Machine Learning","author":"P. Stone","year":"2001","unstructured":"Stone, P., and Sutton, R. S. 2001. Scaling reinforcement learning toward RoboCup soccer. In C. Brodley and A. Danyluk, editors, Proceedings of the Eighteenth International Conference on Machine Learning. San Francisco, CA: Morgan Kaufmann, pp. 537\u2013544."},{"key":"5110821_CR69","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1023\/A:1007510522680","volume":"33","author":"T. Sugawara","year":"1998","unstructured":"Sugawara, T. and Lesser. V. 1998. Learning to improve coordinated actions in cooperative distributed problem-solving environments. Machine Learning 33: 129\u2013154.","journal-title":"Machine Learning"},{"key":"5110821_CR70","first-page":"1038","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1995 Conference","author":"R. S. Sutton","year":"1996","unstructured":"Sutton, R. S. 1996. Generalization in reinforcement learning: Successful examples using sparse coarse coding. In D. S. Touretzky, M. C. Mozer, and M. E. Hasselmo, editors, Advances in Neural Information Processing Systems: Proceedings of the 1995 Conference. Cambridge, MA: MIT Press, pp. 1038\u20131044."},{"key":"5110821_CR71","doi-asserted-by":"crossref","first-page":"135","DOI":"10.1037\/0033-295X.88.2.135","volume":"88","author":"R. S. Sutton","year":"1981","unstructured":"Sutton, R. S., and Barto, A. G. 1981. Toward a modern theory of adaptive networks: Expectation and prediction. Psychological Review 88: 135\u2013170.","journal-title":"Psychological Review"},{"key":"5110821_CR72","volume-title":"Reinforcement Learning: An Introduction","author":"R. S. Sutton","year":"1998","unstructured":"Sutton, R. S., and Barto, A. G. 1998. Reinforcement Learning: An Introduction. Cambridge, MA: MIT Press."},{"key":"5110821_CR73","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"R. S. Sutton","year":"1999","unstructured":"Sutton, R. S., Precup, D., and Singh, S. 1999. Between mdps and semi-mdps: A framework for temporal abstraction in reinforcement learning. Artificial Intelligence 112: 181\u2013211.","journal-title":"Artificial Intelligence"},{"key":"5110821_CR74","doi-asserted-by":"crossref","unstructured":"Tan, M. 1993. Multi-agent reinforcement learning: Independent vs. cooperative agents. In Proceedings of the Tenth International Conference on Machine Learning, pp. 330\u2013337.","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"5110821_CR75","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1023\/A:1022624705476","volume":"8","author":"G. J. Tesauro","year":"1992","unstructured":"Tesauro, G. J. 1992. Practical issues in temporal difference learning. Machine Learning 8: 257\u2013277.","journal-title":"Machine Learning"},{"issue":"2","key":"5110821_CR76","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1162\/neco.1994.6.2.215","volume":"6","author":"G. J. Tesauro","year":"1994","unstructured":"Tesauro, G. J. 1994. TD-gammon, a self-teaching backgammon program, achieves master-level play. Neural Computation 6(2): 215\u2013219.","journal-title":"Neural Computation"},{"key":"5110821_CR77","unstructured":"Theocharous, G. 2002. Hierarchical Learning and Planning in Partially Observable Markov Decision Processes. Ph.D. Thesis, Michigan State University."},{"key":"5110821_CR78","unstructured":"Theocharous, G., and Mahadevan, S. 2002. Approximate planning with hierarchical partially observable Markov decision process for robot navigation. In Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)."},{"key":"5110821_CR79","unstructured":"Theocharous, G., Rohanimanesh, K., and Mahadevan, S. 2001. Learning hierarchical partially observable markov decision processes for robot navigation. In Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)."},{"key":"5110821_CR80","first-page":"385","volume-title":"Advances in Neural Information Processing Systems: Proceedings of the 1994 Conference","author":"S. B. Thrun","year":"1995","unstructured":"Thrun, S. B., and Schwartz, A. 1995. Finding structure in reinforcement learning. In G. Tesauro, D. S. Touretzky, and T. Leen, editors, Advances in Neural Information Processing Systems: Proceedings of the 1994 Conference. Cambridge MA: MIT Press, pp. 385\u2013392."},{"key":"5110821_CR81","first-page":"674","volume":"42","author":"J. N. Tsitsiklis","year":"1997","unstructured":"Tsitsiklis, J. N., and Van Roy, B. 1997. An analysis of temporal-difference learning with function approximation. IEEE Transactions on Automatic Control 42: 674\u2013690.","journal-title":"An analysis of temporal-difference learning with function approximation. IEEE Transactions on Automatic Control"},{"key":"5110821_CR82","volume-title":"Learning from Delayed Rewards","author":"C. J. C. H. Watkins","year":"1989","unstructured":"Watkins, C. J. C. H. 1989. Learning from Delayed Rewards. Ph.D. thesis. Cambridge, U.K.: Cambridge University."},{"key":"5110821_CR83","first-page":"279","volume":"8","author":"C. J. C. H. Watkins","year":"1992","unstructured":"Watkins, C. J. C. H., and Dayan, P. 1992. Q-learning. Machine Learning 8: 279\u2013292.","journal-title":"Machine Learning"},{"key":"5110821_CR84","volume-title":"Multiagent Systems: A Modern Approach to Distributed Artificial Intelligence","author":"G. Weiss","year":"1999","unstructured":"Weiss, G. 1999. Multiagent Systems: A Modern Approach to Distributed Artificial Intelligence. Cambridge, MA: MIT Press."},{"key":"5110821_CR85","first-page":"25","volume":"22","author":"P. J. Werbos","year":"1977","unstructured":"Werbos, P. J. 1977. Advanced forecasting methods for global crisis warning and models of intelligence. General Systems Yearbook 22: 25\u201338.","journal-title":"General Systems Yearbook"},{"key":"5110821_CR86","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/TSMC.1987.289329","volume":"17","author":"P. J. Werbos","year":"1987","unstructured":"Werbos, P. J. 1987. Building and understanding adaptive systems: A statistical\/numerical approach to factory automation and brain research. IEEE Transactions on Systems, Man, and Cybernetics 17: 7\u201320.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics"},{"key":"5110821_CR87","first-page":"493","volume-title":"Handbook of Intelligent Control: Neural, Fuzzy, and Adaptive Approaches","author":"P. J. Werbos","year":"1992","unstructured":"Werbos, P. J. 1992. Approximate dynamic programming for real-time control and neural modeling. In D. A. White and D. A. Sofge, editors, Handbook of Intelligent Control: Neural, Fuzzy, and Adaptive Approaches. New York: Van Nostrand Reinhold, pp. 493\u2013525."},{"key":"5110821_CR88","doi-asserted-by":"crossref","first-page":"591","DOI":"10.1145\/355598.362773","volume":"13","author":"W. A. Woods","year":"1970","unstructured":"Woods, W. A. 1970. Transition network grammars for natural language analysis. Communications of the ACM 13: 591\u2013606.","journal-title":"Communications of the ACM"}],"container-title":["Discrete Event Dynamic Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1022140919877.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1023\/A:1022140919877\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1022140919877.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,29]],"date-time":"2025-07-29T04:11:54Z","timestamp":1753762314000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1023\/A:1022140919877"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2003,1]]},"references-count":88,"journal-issue":{"issue":"1-2","published-print":{"date-parts":[[2003,1]]}},"alternative-id":["5110821"],"URL":"https:\/\/doi.org\/10.1023\/a:1022140919877","relation":{},"ISSN":["0924-6703","1573-7594"],"issn-type":[{"value":"0924-6703","type":"print"},{"value":"1573-7594","type":"electronic"}],"subject":[],"published":{"date-parts":[[2003,1]]}}}