{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T19:47:59Z","timestamp":1725565679771},"publisher-location":"Berlin, Heidelberg","reference-count":8,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540224181"},{"type":"electronic","value":"9783540277729"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2004]]},"DOI":"10.1007\/978-3-540-27772-9_64","type":"book-chapter","created":{"date-parts":[[2010,9,15]],"date-time":"2010-09-15T21:37:11Z","timestamp":1284586631000},"page":"628-633","source":"Crossref","is-referenced-by-count":0,"title":["Merging Individually Learned Optimal Results to Accelerate Coordination"],"prefix":"10.1007","author":[{"given":"Huaxiang","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangteng","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"64_CR1","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"T.G. Dietterich","year":"2000","unstructured":"Dietterich, T.G.: Hierarchical Reinforcement Learning With the MAXQ Value Function Decomposition. J. of Artificial Intelligence Research\u00a013, 227\u2013303 (2000)","journal-title":"J. of Artificial Intelligence Research"},{"key":"64_CR2","unstructured":"Watkins, C.J.C.H.: Learning from Delayed Rewards. Cambridge University. Ph.D. thesis, Cambridge, UK (1989)"},{"key":"64_CR3","unstructured":"Singh, S., Cohn, D.: How to Dynamically Merge Markov Decision Processes. In: 17th Int. Conference on Neural Information Processing Systems (1999)"},{"key":"64_CR4","doi-asserted-by":"crossref","unstructured":"Ghavamzadeh, M., Mahadevan, S.: A Multiagent Reinforcement Learning Algorithm by Dynamically Merging Markov Decision Processes. In: 1st Int. Joint Conference on Autonomous Agents and Multiagent Systems, Bologna (2002)","DOI":"10.1145\/544862.544940"},{"key":"64_CR5","unstructured":"Boutilier, C.: Sequential Optimality and Coordination in Multiagent Systems. In: 16th Int. Joint Conference on Artificial Intelligence, Stockholm, pp. 478\u2013485 (1999)"},{"key":"64_CR6","doi-asserted-by":"crossref","unstructured":"Littman, M.L.: Markov Games as a Framework for Multi-Agent Reinforcement Learning. In: 11th Int. Conference of Machine Learning, New Brunswick, pp. 157\u2013163 (1994)","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"64_CR7","first-page":"1","volume":"1","author":"J. Hu","year":"2003","unstructured":"Hu, J., Wellman, M.P.: Nash Q-Learning for General-Sum Stochastic Games. J. of Machine Learning Research\u00a01, 1\u201330 (2003)","journal-title":"J. of Machine Learning Research"},{"key":"64_CR8","unstructured":"Greenwald, A., Hall, K., Serrano, R.: Correlated-Q Learning. In: 20th Int. Conference on Neural Information Processing Systems, Workshop on Multiagent Learning (2002)"}],"container-title":["Lecture Notes in Computer Science","Advances in Web-Age Information Management"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-27772-9_64.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,3]],"date-time":"2021-05-03T03:24:06Z","timestamp":1620012246000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-27772-9_64"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2004]]},"ISBN":["9783540224181","9783540277729"],"references-count":8,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-27772-9_64","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2004]]}}}