{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T10:05:46Z","timestamp":1769940346241,"version":"3.49.0"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2019,4,27]],"date-time":"2019-04-27T00:00:00Z","timestamp":1556323200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"Honda Research Institute","award":["124232"],"award-info":[{"award-number":["124232"]}]},{"name":"National Science Foundation Graduate Research Fellowship Program","award":["DGE-1656518"],"award-info":[{"award-number":["DGE-1656518"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1007\/s10458-019-09407-z","type":"journal-article","created":{"date-parts":[[2019,4,27]],"date-time":"2019-04-27T08:12:39Z","timestamp":1556352759000},"page":"330-352","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Decomposition methods with deep corrections for reinforcement learning"],"prefix":"10.1007","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9151-1513","authenticated-orcid":false,"given":"Maxime","family":"Bouton","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kyle D.","family":"Julian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alireza","family":"Nakhaei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kikuo","family":"Fujimura","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mykel J.","family":"Kochenderfer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,4,27]]},"reference":[{"key":"9407_CR1","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/10187.001.0001","volume-title":"Decision making under uncertainty: Theory and application","author":"MJ Kochenderfer","year":"2015","unstructured":"Kochenderfer, M. J. (2015). Decision making under uncertainty: Theory and application. Cambridge: MIT Press."},{"key":"9407_CR2","unstructured":"Russell, S. J., & Zimdars, A. (2003). Q-decomposition for reinforcement learning agents. In International conference on machine learning (ICML)."},{"key":"9407_CR3","unstructured":"Tesauro, G. (2005). Online resource allocation using decompositional reinforcement learning. In AAAI conference on artificial intelligence (AAAI)."},{"key":"9407_CR4","unstructured":"Bernstein, D. S., Zilberstein, S., & Immerman, N. (2000). The complexity of decentralized control of Markov decision processes. In Conference on uncertainty in artificial intelligence (UAI)."},{"issue":"1","key":"9407_CR5","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1023\/A:1008916000526","volume":"9","author":"JK Rosenblatt","year":"2000","unstructured":"Rosenblatt, J. K. (2000). Optimal selection of uncertain actions by maximizing expected utility. Autonomous Robots, 9(1), 17\u201325.","journal-title":"Autonomous Robots"},{"issue":"2","key":"9407_CR6","doi-asserted-by":"publisher","first-page":"398","DOI":"10.2514\/1.54805","volume":"35","author":"JP Chryssanthacopoulos","year":"2012","unstructured":"Chryssanthacopoulos, J. P., & Kochenderfer, M. J. (2012). Decomposition methods for optimized collision avoidance with multiple threats. AIAA Journal of Guidance, Control, and Dynamics, 35(2), 398\u2013405.","journal-title":"AIAA Journal of Guidance, Control, and Dynamics"},{"key":"9407_CR7","doi-asserted-by":"crossref","unstructured":"Ong, H. Y., & Kochenderfer, M. J. (2015). Short-term conflict resolution for unmanned aircraft traffic management. In Digital avionics systems conference (DASC).","DOI":"10.1109\/DASC.2015.7311424"},{"key":"9407_CR8","doi-asserted-by":"crossref","unstructured":"Wray, K. H., Witwicki, S. J., & Zilberstein, S. (2017). Online decision-making for scalable autonomous systems. In International joint conference on artificial intelligence (IJCAI).","DOI":"10.24963\/ijcai.2017\/664"},{"issue":"1","key":"9407_CR9","doi-asserted-by":"publisher","first-page":"186","DOI":"10.1109\/TCYB.2015.2509646","volume":"47","author":"S Hung","year":"2017","unstructured":"Hung, S., & Givigi, S. N. (2017). A q-learning approach to flocking with UAVs in a stochastic environment. IEEE Transactions on Cybernetics, 47(1), 186\u2013197.","journal-title":"IEEE Transactions on Cybernetics"},{"key":"9407_CR10","doi-asserted-by":"crossref","unstructured":"Tan, M. (1993). Multi-agent reinforcement learning: Independent versus cooperative agents. In International conference on machine learning (ICML).","DOI":"10.1023\/A:1022679428250"},{"key":"9407_CR11","unstructured":"Claus, C., & Boutilier, C. (1998). The dynamics of reinforcement learning in cooperative multiagent systems. In AAAI conference on artificial intelligence (AAAI)."},{"key":"9407_CR12","doi-asserted-by":"crossref","unstructured":"Tompa, R. E., & Kochenderfer, M. J. (2018). Optimal aircraft rerouting during space launches using adaptive spatial discretization. In Digital avionics systems conference (DASC).","DOI":"10.1109\/DASC.2018.8569888"},{"key":"9407_CR13","unstructured":"Van der Pol, E., & Oliehoek, F. A. (2016). Coordinated deep reinforcement learners for traffic light control. In NIPS workshop on learning, inference and control of multi-agent systems."},{"key":"9407_CR14","doi-asserted-by":"crossref","unstructured":"Julian, K. D., & Kochenderfer, M. J. (2018). Autonomous distributed wildfire surveillance using deep reinforcement learning. In AIAA guidance, navigation, and control conference (GNC), decomposition methods with deep corrections for reinforcement learning 25.","DOI":"10.2514\/6.2018-1589"},{"key":"9407_CR15","unstructured":"Oliehoek, F. A., Whiteson, S., & Spaan, M. T. J. (2013). Approximate solutions for factored Dec-POMDPs with many agents. In International conference on autonomous agents and multiagent systems (AAMAS)."},{"key":"9407_CR16","unstructured":"Sunehag, P., Lever, G., Gruslys, A., Czarnecki, W. M., Zambaldi, V. F., Jaderberg, M., et al. (2018). Value-decomposition networks for cooperative multi-agent learning based on team reward. In International conference on autonomous agents and multi-agent systems (AAMAS)."},{"issue":"7540","key":"9407_CR17","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A. A., Veness, J., Bellemare, M. G., et al. (2015). Human-level control through deep reinforcement learning. Nature, 518(7540), 529\u2013533.","journal-title":"Nature"},{"key":"9407_CR18","doi-asserted-by":"crossref","unstructured":"Gu, S., Holly, E., Lillicrap, T. P., & Levine, S. (2017). Deep reinforcement learning for robotic manipulation with asynchronous off-policy updates. In IEEE international conference on robotics and automation (ICRA).","DOI":"10.1109\/ICRA.2017.7989385"},{"key":"9407_CR19","doi-asserted-by":"crossref","unstructured":"Smart, W. D., & Kaelbling, L. P. (2002). Effective reinforcement learning for mobile robots. In IEEE international conference on robotics and automation (ICRA).","DOI":"10.1109\/ROBOT.2002.1014237"},{"key":"9407_CR20","doi-asserted-by":"crossref","unstructured":"Gottwald, M., Meyer, D., Shen, H., & Diepold, K. (2017). Learning to walk with prior knowledge. In IEEE international conference on advanced intelligent mechatronics (AIM).","DOI":"10.1109\/AIM.2017.8014209"},{"issue":"3","key":"9407_CR21","doi-asserted-by":"publisher","first-page":"655","DOI":"10.1109\/TRO.2015.2419431","volume":"31","author":"M Cutler","year":"2015","unstructured":"Cutler, M., Walsh, T. J., & How, J. P. (2015). Real-world reinforcement learning via multifidelity simulators. IEEE Transactions on Robotics, 31(3), 655\u2013671.","journal-title":"IEEE Transactions on Robotics"},{"key":"9407_CR22","unstructured":"Schaul, T., Quan, J., Antonoglou, I., & Silver, D. (2016). Prioritized experience replay. In International conference on learning representations (ICLR)."},{"key":"9407_CR23","doi-asserted-by":"crossref","unstructured":"Van Hasselt, H., Guez, A., & Silver, D. (2016). Deep reinforcement learning with double q-learning. In AAAI conference on artificial intelligence (AAAI).","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"9407_CR24","unstructured":"Wang, Z., Schaul, T., Hessel, M., van Hasselt, H., Lanctot, M., & de Freitas, N. (2016). Dueling network architectures for deep reinforcement learning. In International conference on machine learning (ICML)."},{"key":"9407_CR25","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192","volume-title":"Reinforcement learning\u2014An introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning\u2014An introduction. Cambridge: MIT Press."},{"key":"9407_CR26","doi-asserted-by":"crossref","unstructured":"Eldred, M., & Dunlavy, D. (2006). Formulations for surrogate-based optimization with data fit, multifidelity, and reduced-order models. In AIAA\/ISSMO multi-disciplinary analysis and optimization conference.","DOI":"10.2514\/6.2006-7117"},{"key":"9407_CR27","doi-asserted-by":"crossref","unstructured":"Rajnarayan, D., Haas, A., & Kroo, I. (2008). A multifidelity gradient-free optimization method and application to aerodynamic design. In AIAA\/ISSMO multidisciplinary analysis and optimization conference.","DOI":"10.2514\/6.2008-6020"},{"issue":"26","key":"9407_CR28","first-page":"1","volume":"18","author":"M Egorov","year":"2017","unstructured":"Egorov, M., Sunberg, Z. N., Balaban, E., Wheeler, T. A., Gupta, J. K., & Kochenderfer, M. J. (2017). POMDPs.jl: A framework for sequential decision making under uncertainty. Journal of Machine Learning, 18(26), 1\u20135.","journal-title":"Journal of Machine Learning"},{"issue":"9","key":"9407_CR29","doi-asserted-by":"publisher","first-page":"1288","DOI":"10.1177\/0278364914528255","volume":"33","author":"H Bai","year":"2014","unstructured":"Bai, H., Hsu, D., & Lee, W. S. (2014). Integrated perception and planning in the continuous space: A POMDP approach. International Journal of Robotics Research, 33(9), 1288\u20131302.","journal-title":"International Journal of Robotics Research"},{"key":"9407_CR30","doi-asserted-by":"crossref","unstructured":"Brechtel, S., Gindele, T., & Dillmann, R. (2014). Probabilistic decision-making under uncertainty for autonomous driving using continuous POMDPs. In IEEE international conference on intelligent transportation systems (ITSC).","DOI":"10.1109\/ITSC.2014.6957722"},{"key":"9407_CR31","unstructured":"Bandyopadhyay, T., Won, K. S., Frazzoli, E., Hsu, D., Lee, W. S., & Rus, D. (2012). Intention-aware motion planning. In Algorithmic foundations of robotics X."},{"key":"9407_CR32","doi-asserted-by":"crossref","unstructured":"Chae, H., Kang, C. M., Kim, B., Kim, J., Chung, C. C., & Choi, J. W. (2017) Autonomous braking system via deep reinforcement learning. In IEEE international conference on intelligent transportation systems (ITSC).","DOI":"10.1109\/ITSC.2017.8317839"},{"key":"9407_CR33","doi-asserted-by":"crossref","unstructured":"Chen, B., Zhao, D., & Peng, H. (2017). Evaluation of automated vehicles encountering pedestrians at unsignalized crossings. In IEEE intelligent vehicles symposium (IV).","DOI":"10.1109\/IVS.2017.7995950"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-019-09407-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10458-019-09407-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-019-09407-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,17]],"date-time":"2022-09-17T00:15:07Z","timestamp":1663373707000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10458-019-09407-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,4,27]]},"references-count":33,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2019,5]]}},"alternative-id":["9407"],"URL":"https:\/\/doi.org\/10.1007\/s10458-019-09407-z","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,4,27]]},"assertion":[{"value":"27 April 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}