{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T09:42:09Z","timestamp":1785577329092,"version":"3.56.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2006,4,4]],"date-time":"2006-04-04T00:00:00Z","timestamp":1144108800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2006,9]]},"DOI":"10.1007\/s10458-006-7035-4","type":"journal-article","created":{"date-parts":[[2006,4,5]],"date-time":"2006-04-05T06:54:38Z","timestamp":1144220078000},"page":"197-229","source":"Crossref","is-referenced-by-count":92,"title":["Hierarchical multi-agent reinforcement learning"],"prefix":"10.1007","volume":"13","author":[{"given":"Mohammad","family":"Ghavamzadeh","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sridhar","family":"Mahadevan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rajbala","family":"Makar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2006,4,4]]},"reference":[{"key":"7035_CR1","unstructured":"Askin R., Standridge C. (1993). Modeling and analysis of manufacturing systems. John Wiley and Sons."},{"key":"7035_CR2","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/70.736776","volume":"14","author":"T. Balch","year":"1998","unstructured":"Balch T. and Arkin R. (1998). Behavior-based formation control for multi-robot Teams. IEEE Transactions on Robotics and Automation, 14: 1\u201315","journal-title":"IEEE Transactions on Robotics and Automation"},{"key":"7035_CR3","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1023\/A:1022140919877","volume":"13","author":"A. Barto","year":"2003","unstructured":"Barto A. and Mahadevan S. (2003). Recent advances in hierarchical reinforcement learning. Discrete Event Systems Special Issue on Reinforcement Learning, 13: 41\u201377","journal-title":"Discrete Event Systems Special Issue on Reinforcement Learning"},{"key":"7035_CR4","unstructured":"Bernstein D., Zilberstein S., Immerman N. (2000). The complexity of decentralized control of markov decision processes. In Proceedings of the sixteenth international conference on uncertainty in artificial intelligence (pp. 32\u201337)."},{"key":"7035_CR5","unstructured":"Boutilier, C. (1999). Sequential optimality coordination in multi-agent systems. In Proceedings of the sixteenth international joint conference on artificial intelligence (pp. 478\u2013485)."},{"key":"7035_CR6","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1016\/S0004-3702(02)00121-2","volume":"136","author":"M. Bowling","year":"2002","unstructured":"Bowling M. and Veloso M. (2002). Multiagent learning using a variable learning rate. Artificial Intelligence, 136: 215\u2013250","journal-title":"Artificial Intelligence"},{"key":"7035_CR7","doi-asserted-by":"crossref","first-page":"451","DOI":"10.1613\/jair.839","volume":"17","author":"H. Bui","year":"2002","unstructured":"Bui H., Venkatesh S. and West G. (2002). Policy recognition in the abstract hidden markov model. Journal of Artificial Intelligence Research, 17: 451\u2013499","journal-title":"Journal of Artificial Intelligence Research"},{"key":"7035_CR8","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1007518724497","volume":"33","author":"R. Crites","year":"1998","unstructured":"Crites R. and Barto A. (1998). Elevator group control using multiple reinforcement learning agents. Machine Learning, 33: 235\u2013262","journal-title":"Machine Learning"},{"key":"7035_CR9","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"T. Dietterich","year":"2000","unstructured":"Dietterich T. (2000). Hierarchical reinforcement learning with the MAXQ value function decomposition. Journal of Artificial Intelligence Research, 13: 227\u2013303","journal-title":"Journal of Artificial Intelligence Research"},{"key":"7035_CR10","unstructured":"Filar, J., Vrieze, K. (1997). Competitive Markov decision processes. Springer Verlag."},{"key":"7035_CR11","unstructured":"Ghavamzadeh, M., Mahadevan, S. (2003). Hierarchical policy gradient algorithms. In Proceedings of the twentieth international conference on machine learning (pp. 226\u2013233)."},{"key":"7035_CR12","unstructured":"Ghavamzadeh, M., Mahadevan, S. (2004). Learning to communicate and act using hierarchical reinforcement learning. In Proceedings of the third international joint conference on autonomous agents and multiagent systems (pp. 1114\u20131121)."},{"key":"7035_CR13","unstructured":"Guestrin, C., Lagoudakis, M., Parr, R. (2002). Coordinated reinforcement learning. In Proceedings of the nineteenth international conference on machine learning (pp. 227\u2013234)."},{"key":"7035_CR14","unstructured":"Howard, R. (1971). Dynamic probabilistic systems: Semi-Markov and decision processes. John Wiley and Sons."},{"key":"7035_CR15","unstructured":"Hu, J., Wellman, M. (1998). Multiagent reinforcement learning: Theoretical framework and an algorithm. In Proceedings of the fifteenth international conference on machine learning (pp. 242\u2013250)."},{"key":"7035_CR16","unstructured":"Kearns, M., Littman, M., Singh, S. (2001). Graphical models for game theory. In Proceedings of the seventeenth international conference on uncertainty in artificial intelligence (pp. 253\u2013260)."},{"key":"7035_CR17","doi-asserted-by":"crossref","first-page":"95","DOI":"10.1080\/00207549608904893","volume":"34","author":"C. Klein","year":"1996","unstructured":"Klein C. and Kim J. (1996). AGV dispatching. International Journal of Production Research, 34: 95\u2013110","journal-title":"International Journal of Production Research"},{"key":"7035_CR18","unstructured":"Koller D., Milch B. Multiagent influence diagrams for representing and solving games. In Proceedings of the seventeenth international joint conference on artificial intelligence (pp. 1027\u20131034)."},{"key":"7035_CR19","unstructured":"La Mura, P. (2000). Game Networks. In Proceedings of the sixteenth international conference on uncertainty in artificial intelligence."},{"key":"7035_CR20","doi-asserted-by":"crossref","first-page":"121","DOI":"10.1177\/003754979606600208","volume":"66","author":"J. Lee","year":"1996","unstructured":"Lee J. (1996). Composite dispatching rules for multiple-vehicle agv systems. Simulation, 66: 121\u2013130","journal-title":"Simulation"},{"key":"7035_CR21","doi-asserted-by":"crossref","unstructured":"Littman, M. (1994). Markov games as a framework for multi-agent reinforcement learning. In Proceedings of the eleventh international conference on machine learning (pp. 157\u2013163).","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"7035_CR22","unstructured":"Littman, M. (2001). Friend-or-Foe Q-learning in general-sum games. In Proceedings of the eighteenth international conference on machine learning (pp. 322\u2013328)."},{"key":"7035_CR23","unstructured":"Littman, M., Kearns, M., Singh, S. (2001). An efficient exact algorithm for singly connected graphical games. In Proceedings of neural information processing systems (pp. 817\u2013824)."},{"key":"7035_CR24","doi-asserted-by":"crossref","unstructured":"Makar, R., Mahadevan, S., Ghavamzadeh, M. (2001). Hierarchical multi-agent reinforcement learning. In Proceedings of the fifth international conference on autonomous agents (pp. 246\u2013253).","DOI":"10.1145\/375735.376302"},{"key":"7035_CR25","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1023\/A:1008819414322","volume":"4","author":"M. Mataric","year":"1997","unstructured":"Mataric M. (1997). Reinforcement learning in the multi-robot domain (1997). Autonomous Robots, 4: 73\u201383","journal-title":"Autonomous Robots"},{"key":"7035_CR26","unstructured":"Ortiz, L., Kearns, M. (2002). Nash propagation for loopy graphical games. In Proceedings of neural information processing systems."},{"key":"7035_CR27","unstructured":"Owen, G. (1995). Game theory. Academic Press."},{"key":"7035_CR28","unstructured":"Parr, R. (1998). Hierarchical control and learning for Markov decision processes, PhD thesis, University of California, Berkeley."},{"key":"7035_CR29","unstructured":"Peshkin, L., Kim, K., Meuleau, N., Kaelbling, L. (2000). Learning to cooperate via policy search. In Proceedings of the sixteenth international conference on uncertainty in artificial intelligence (pp. 489\u2013496)."},{"key":"7035_CR30","doi-asserted-by":"crossref","unstructured":"Puterman, M. (1994). Markov decision processes. Wiley Interscience.","DOI":"10.1002\/9780470316887"},{"key":"7035_CR31","doi-asserted-by":"crossref","first-page":"389","DOI":"10.1613\/jair.1024","volume":"16","author":"D. Pynadath","year":"2002","unstructured":"Pynadath D. and Tambe M. (2002). The communicative multiagent team decision problem: Analyzing teamwork theories and models. Journal of Artificial Intelligence Research, 16: 389\u2013426","journal-title":"Journal of Artificial Intelligence Research"},{"key":"7035_CR32","unstructured":"Rohanimanesh, K., Mahadevan, S. Learning to take concurrent actions. In Proceedings of the sixteenth annual conference on neural information processing systems."},{"key":"7035_CR33","unstructured":"Saria, S., Mahadevan, M. (2004). Probabilistic plan recognition in multiagent systems. In Proceedings of the fourteenth international conference on automated planning and scheduling (pp. 12\u201322)."},{"key":"7035_CR34","unstructured":"Schneider, J., Wong, W., Moore, A., Riedmiller, M. Distributed value functions. In Proceedings of the sixteenth international conference on machine learning (pp. 371\u2013378)."},{"key":"7035_CR35","unstructured":"Singh, S., Kearns, M., Mansour, Y. (2000). Nash convergence of gradient dynamics in general-sum games. In Proceedings of the sixteenth international conference on uncertainty in artificial intelligence (pp. 541\u2013548)."},{"key":"7035_CR36","doi-asserted-by":"crossref","unstructured":"Stone, P., Veloso, M. (1999). Team-partitioned, opaque-transition reinforcement learning. In Proceedings of the third international conference on autonomous agents (pp. 206\u2013212).","DOI":"10.1145\/301136.301195"},{"key":"7035_CR37","doi-asserted-by":"crossref","unstructured":"Sugawara T., Lesser V. Learning to improve coordinated actions in cooperative distributed problem-solving environments. Machine Learning, 33: 129\u2013154.","DOI":"10.1023\/A:1007510522680"},{"key":"7035_CR38","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"R. Sutton","year":"1999","unstructured":"Sutton R., Precup D. and Singh S. (1999). Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning. Artificial Intelligence, 112: 181\u2013211","journal-title":"Artificial Intelligence"},{"key":"7035_CR39","unstructured":"Tadepalli, P., Ok, D. Scaling up average reward reinforcement learning by approximating the domain models and the value function. In Proceedings of the thirteenth international conference on machine learning (pp. 471\u2013479)."},{"key":"7035_CR40","doi-asserted-by":"crossref","unstructured":"Tan, M. (1993). Multi-agent reinforcement learning: Independent vs. cooperative agents. In Proceedings of the tenth international conference on machine learning (pp. 330\u2013337).","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"7035_CR41","unstructured":"Vickrey, D., Koller, D. (2002). Multiagent algorithms for solving graphical games. In Proceedings of the national conference on artificial intelligence (pp. 345\u2013351)."},{"key":"7035_CR42","unstructured":"Watkins, C. (1989). Learning from delayed rewards, PhD thesis, Kings College, Cambridge, England."},{"key":"7035_CR43","unstructured":"Weiss, G. (1999). Multi-agent systems: A modern approach to distributed artificial intelligence. MIT Press."},{"key":"7035_CR44","doi-asserted-by":"crossref","unstructured":"Xuan, P., Lesser, V., Zilberstein, S. (2001). Communication decisions in multi-agent cooperation: Model and experiments. In Proceedings of the fifth international conference on autonomous agents (pp. 616\u2013623).","DOI":"10.1145\/375735.376469"},{"key":"7035_CR45","unstructured":"Xuan, P., Lesser, V. (2002). Multiagent policies: From centralized ones to decentralized ones. In Proceedings of the first international joint conference on autonomous agents and multiagent systems (pp. 1098\u20131105)."}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-006-7035-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10458-006-7035-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-006-7035-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,29]],"date-time":"2019-05-29T17:28:22Z","timestamp":1559150902000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10458-006-7035-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2006,4,4]]},"references-count":45,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2006,9]]}},"alternative-id":["7035"],"URL":"https:\/\/doi.org\/10.1007\/s10458-006-7035-4","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2006,4,4]]}}}