{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T20:54:50Z","timestamp":1774385690252,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":20,"publisher":"ACM","license":[{"start":{"date-parts":[[2005,7,25]],"date-time":"2005-07-25T00:00:00Z","timestamp":1122249600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2005,7,25]]},"DOI":"10.1145\/1082473.1082483","type":"proceedings-article","created":{"date-parts":[[2005,11,7]],"date-time":"2005-11-07T12:34:39Z","timestamp":1131366879000},"page":"60-66","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Efficient learning of multi-step best response"],"prefix":"10.1145","author":[{"given":"Bikramjit","family":"Banerjee","sequence":"first","affiliation":[{"name":"Tulane University, New Orleans, LA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Peng","sequence":"additional","affiliation":[{"name":"Tulane University, New Orleans, LA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2005,7,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"The Evolution of Cooperation","author":"Axelrod R.","year":"1984","unstructured":"R. Axelrod. The Evolution of Cooperation. Basic Books, 1984."},{"key":"e_1_3_2_1_2_1","volume-title":"Advances in Neural Information Processing Systems 16","author":"Bagnell J. A.","year":"2003","unstructured":"J. A. Bagnell, S. Kakade, A. Y. Ng, and J. Schneider. Policy search by dynamic programming. In Advances in Neural Information Processing Systems 16. MIT Press, 2003."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.5555\/1597148.1597151"},{"key":"e_1_3_2_1_4_1","volume-title":"Athena Scientific","author":"Bertsekas D. P.","year":"1995","unstructured":"D. P. Bertsekas. Dynamic Programming and Optimal Control. Athena Scientific, Belmont, MA, 1995."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(02)00121-2"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1162\/153244303765208377"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/1892875.1892884"},{"key":"e_1_3_2_1_8_1","first-page":"64","volume-title":"Proceedings of the Third International Conference on Multiagent Systems","author":"Carmel D.","year":"1998","unstructured":"D. Carmel and S. Markovitch. How to explore your opponent's stratey (almost) optimally. In Proceedings of the Third International Conference on Multiagent Systems, pages 64--71, Los Alamitos, CA, 1998. IEEE Computer Society."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1080\/095281398146789"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 20th International Conference on Machine Learning","author":"Conitzer V.","year":"2003","unstructured":"V. Conitzer and T. Sandholm. AWESOME: A general multiagent learning algorithm that converges in self-play and learns a best response against stationary opponents. In Proceedings of the 20th International Conference on Machine Learning, 2003."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/180139.181019"},{"key":"e_1_3_2_1_12_1","first-page":"116","volume-title":"Proceedings of the 14th International Conference on Machine Learning","author":"Fiechter C.","year":"1997","unstructured":"C. Fiechter. Expected mistake bound model for on-line reinforcement learning. In Proceedings of the 14th International Conference on Machine Learning, pages 116 -- 124, 1997."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/195058.195448"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.5555\/795662.796296"},{"key":"e_1_3_2_1_15_1","unstructured":"S. Kakade. On the Sample Complexity of Reinforcement Learning. PhD thesis University College London 2003."},{"key":"e_1_3_2_1_16_1","volume-title":"Advances in Neural Information Processing Systems 12","author":"Kearns M.","year":"2000","unstructured":"M. Kearns, Y. Mansour, and A. Ng. Approximate planning in large pomdps via reusable trajectories. In Advances in Neural Information Processing Systems 12. MIT Press, 2000."},{"key":"e_1_3_2_1_17_1","first-page":"260","volume-title":"Proceedings of the 15th International Conference on Machine Learning","author":"Kearns M.","year":"1998","unstructured":"M. Kearns and S. Singh. Near-optimal reinforcement learning in polynomial time. In Proceedings of the 15th International Conference on Machine Learning, pages 260 -- 268. Morgan Kaufmann, 1998."},{"key":"e_1_3_2_1_18_1","volume-title":"Proc. National Conf. on Artificial Intelligence","author":"Pivazyan K.","year":"2002","unstructured":"K. Pivazyan and Y. Shoham. Polynomial-time reinforcement learning of near-optimal policies. In Proc. National Conf. on Artificial Intelligence, 2002."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/2073946.2074009"},{"key":"e_1_3_2_1_20_1","volume-title":"Advances in Neural Information Processing Systems 16","author":"Tesauro G.","year":"2004","unstructured":"G. Tesauro. Extending Q-learning to general adaptive multi-agent systems. In S. Thrun, L. Saul, and B. Sch\u00f6lkopf, editors, Advances in Neural Information Processing Systems 16. MIT Press, Cambridge, MA, 2004."}],"event":{"name":"AAMAS05: AAMAS '05 - Fourth International Joint Conference on Autonomous Agents and Multiagent Systems 2005","location":"The Netherlands","acronym":"AAMAS05","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence"]},"container-title":["Proceedings of the fourth international joint conference on Autonomous agents and multiagent systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1082473.1082483","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/1082473.1082483","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T17:26:51Z","timestamp":1774373211000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1082473.1082483"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2005,7,25]]},"references-count":20,"alternative-id":["10.1145\/1082473.1082483","10.1145\/1082473"],"URL":"https:\/\/doi.org\/10.1145\/1082473.1082483","relation":{},"subject":[],"published":{"date-parts":[[2005,7,25]]},"assertion":[{"value":"2005-07-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}