{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T15:53:39Z","timestamp":1725551619148},"publisher-location":"Berlin, Heidelberg","reference-count":17,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540304623"},{"type":"electronic","value":"9783540316527"}],"license":[{"start":{"date-parts":[[2005,1,1]],"date-time":"2005-01-01T00:00:00Z","timestamp":1104537600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2005]]},"DOI":"10.1007\/11589990_71","type":"book-chapter","created":{"date-parts":[[2005,11,26]],"date-time":"2005-11-26T06:28:03Z","timestamp":1132986483000},"page":"684-694","source":"Crossref","is-referenced-by-count":1,"title":["N-Learning: A Reinforcement Learning Paradigm for Multiagent Systems"],"prefix":"10.1007","author":[{"given":"Mark","family":"Mansfield","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J. J.","family":"Collins","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Malachy","family":"Eaton","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Thomas","family":"Collins","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"71_CR1","volume-title":"Handbook of Learing and Approximate Dynamic Programming","author":"A.G. Barto","year":"2004","unstructured":"Barto, A.G., Dietterich, T.G.: Reinforcement learning and its relationship to supervised learning. In: Handbook of Learing and Approximate Dynamic Programming. Wiley-IEEE Press, Cambridge (2004)"},{"issue":"3","key":"71_CR2","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1023\/A:1007634325138","volume":"40","author":"J. Baxter","year":"2000","unstructured":"Baxter, J., Tridgell, A., Weaver, L.: Learning to play chess using temporal differences. Mach. Learn.\u00a040(3), 243\u2013263 (2000)","journal-title":"Mach. Learn."},{"key":"71_CR3","unstructured":"Crites, R.H., Barto, A.G.: Improving elevator performance using reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a08. The MIT Press, Cambridge"},{"key":"71_CR4","unstructured":"Crook, P., Hayes, G.: Learning in a State of confusion: Perceptual aliasing in grid world navigation. In: Towards Intelligent Mobile Robots 2003 (TIMR 2003), 4 British Conference on (Mobile) Robotics, UWE, Bristol (2003)"},{"key":"71_CR5","doi-asserted-by":"crossref","unstructured":"Ficici, S.G., Pollack, J.B.: Statistical reasoning strategies in the pursuit and evasion domain. In: European Conference on Artificial Life, pp. 79\u201388 (1999)","DOI":"10.1007\/3-540-48304-7_13"},{"key":"71_CR6","unstructured":"Miller, G.F., Cliff, D.: Co-evolution of pursuit and evasion I: Biological and game-theoretic fouondations. Technical Report CSRP311 (1994)"},{"key":"71_CR7","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1109\/ICMAS.2000.858456","volume-title":"ICMAS 2000: Proceedings of the Fourth International Conference on MultiAgent Systems (ICMAS-2000)","author":"Y. Nagayuki","year":"2000","unstructured":"Nagayuki, Y., Ishii, S., Doya, K.: Multi-agent reinforcement learning: An approach based on the other agent\u2019s internal model. In: ICMAS 2000: Proceedings of the Fourth International Conference on MultiAgent Systems (ICMAS-2000), Washington, DC, USA, p. 215. IEEE Computer Society, Los Alamitos (2000)"},{"key":"71_CR8","doi-asserted-by":"crossref","first-page":"618","DOI":"10.7551\/mitpress\/3118.003.0074","volume-title":"From animals to animats 4","author":"N. Ono","year":"1996","unstructured":"Ono, N., Fukumoto, K., Ikeda, O.: Collective behavior by modular reinforcement learning animats. In: From animals to animats 4, pp. 618\u2013624. MIT Press, Cambridge (1996)"},{"key":"71_CR9","first-page":"3404","volume-title":"Proceedings of the IEEE International Conference on Robotics and Automation","author":"W. Smart","year":"2002","unstructured":"Smart, W., Kaelbling, L.: Effective reinforcement learning for mobile robots. In: Proceedings of the IEEE International Conference on Robotics and Automation, pp. 3404\u20133410. IEEE, Piscataway (2002)"},{"key":"71_CR10","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1007\/3-540-48422-1_21","volume-title":"RoboCup-98: Robot Soccer World Cup II","author":"P. Stone","year":"1999","unstructured":"Stone, P., Veloso, M.M.: Team-partitioned, opaque-transition reinforced learning. In: Asada, M., Kitano, H. (eds.) RoboCup 1998. LNCS (LNAI), vol.\u00a01604, pp. 261\u2013272. Springer, Heidelberg (1999)"},{"key":"71_CR11","first-page":"353","volume-title":"Proceedings of the Eighth International Workshop on Machine Learning","author":"R.S. Sutton","year":"1991","unstructured":"Sutton, R.S.: Planning by incremental dynamic programming. In: Proceedings of the Eighth International Workshop on Machine Learning, pp. 353\u2013357. Morgan Kaufmann, San Francisco (1991)"},{"issue":"3","key":"71_CR12","doi-asserted-by":"publisher","first-page":"58","DOI":"10.1145\/203330.203343","volume":"38","author":"G. Tesauro","year":"1995","unstructured":"Tesauro, G.: Temporal difference learning and td-gammon. Commun. ACM\u00a038(3), 58\u201368 (1995)","journal-title":"Commun. ACM"},{"key":"71_CR13","unstructured":"Watkins, C.J.: Learning from delayed rewards. PhD thesis, University of Cambridge, Cambridge, England (1989)"},{"key":"71_CR14","unstructured":"Wei\u00df, G.: Technical report fki-233-90: A multiagent framework for planning, reacting and learning. Technical report, D-80290 Munchen, Germany (1999)"},{"key":"71_CR15","volume-title":"Robot Learning","author":"S. Whitehead","year":"1993","unstructured":"Whitehead, S.: Learning multiple goal behavior via task decomposition and dynamic policy merging. In: Connell, J.H., Mahadevan, S. (eds.) Robot Learning. Kluwer Academic Publishers, Norwell (1993)"},{"issue":"1","key":"71_CR16","first-page":"45","volume":"7","author":"S.D. Whitehead","year":"1991","unstructured":"Whitehead, S.D., Ballard, D.H.: Learning to perceive and act by trial and error. Mach. Learn.\u00a07(1), 45\u201383 (1991)","journal-title":"Mach. Learn."},{"key":"71_CR17","doi-asserted-by":"crossref","first-page":"516","DOI":"10.7551\/mitpress\/3118.003.0062","volume-title":"Proceedings of the Fourth International Conference on Simulation of Adaptive Behavior: From animals to animats 4","author":"J. Zhao","year":"1996","unstructured":"Zhao, J., Schmidhuber, J.: Incremental self-improvement for life-time multi-agent reinforcement learning. In: Maes, P., Mataric, M.J., Meyer, J.-A., Pollack, J., Wilson, S.W. (eds.) Proceedings of the Fourth International Conference on Simulation of Adaptive Behavior: From animals to animats 4, Cape Code, USA, pp. 516\u2013525. MIT Press, Cambridge (1996)"}],"container-title":["Lecture Notes in Computer Science","AI 2005: Advances in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/11589990_71","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T04:17:25Z","timestamp":1706761045000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/11589990_71"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2005]]},"ISBN":["9783540304623","9783540316527"],"references-count":17,"URL":"https:\/\/doi.org\/10.1007\/11589990_71","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2005]]}}}