{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T07:16:00Z","timestamp":1781248560176,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2006,5,8]],"date-time":"2006-05-08T00:00:00Z","timestamp":1147046400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2006,5,8]]},"DOI":"10.1145\/1160633.1160762","type":"proceedings-article","created":{"date-parts":[[2006,10,18]],"date-time":"2006-10-18T18:04:00Z","timestamp":1161194640000},"page":"720-727","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":142,"title":["Probabilistic policy reuse in a reinforcement learning agent"],"prefix":"10.1145","author":[{"given":"Fernando","family":"Fern\u00e1ndez","sequence":"first","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manuela","family":"Veloso","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2006,5,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of IJCAI-99","author":"Bowling M.","year":"1999","unstructured":"M. Bowling and M. Veloso. Bounding the suboptimality of reusing subproblems. In Proceedings of IJCAI-99, 1999."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the International Conference on Machine Learning and Applications","author":"Carroll J.","year":"2002","unstructured":"J. Carroll and T. Peterson. Fixed vs. dynamic sub-transfer in reinforcement learning. In Proceedings of the International Conference on Machine Learning and Applications, 2002."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622262.1622268"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the European Conference on Artificial Intelligence (ECAI 2002","author":"Fern\u00e1ndez F.","year":"2002","unstructured":"F. Fern\u00e1ndez and D. Borrajo. On determinism handling while learning reduced state space representations. In Proceedings of the European Conference on Artificial Intelligence (ECAI 2002), Lyon (France), July 2002."},{"key":"e_1_3_2_1_5_1","unstructured":"F. Fern\u00e1ndez and M. Veloso. Exploration and policy reuse. Technical Report CMU-CS-05-172 School of Computer Science Carnegie Mellon University 2005."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/645531.656017"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Twentieth National Conference on Artificial Intelligence","author":"Maclin R.","year":"2005","unstructured":"R. Maclin, J. Shavlik, L. Torrey, T. Walker, and E. Wild. Giving advice about preferred actions to reinforcement learners via knowledge-based kernel regression. In Proceedings of the Twentieth National Conference on Artificial Intelligence, July 2005. To appear."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:AIRE.0000036264.95672.64"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1002\/int.20105"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622434.1622450"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the Twentieth National Conference on Artificial Intelligence","author":"Sherstov A. A.","year":"2005","unstructured":"A. A. Sherstov and P. Stone. Improving action selection in MDP's via knowledge transfer. In Proceedings of the Twentieth National Conference on Artificial Intelligence, July 2005."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102454"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the Internacional Conference on Machine Learning (ICML'98)","author":"Sutton R. S.","year":"1998","unstructured":"R. S. Sutton, D. Precup, and S. Singh. Intra-option learning about temporally abstract actions. In Proceedings of the Internacional Conference on Machine Learning (ICML'98), 1998."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1082473.1082482"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the Twentieth National Conference on Artificial Intelligence","author":"Taylor M. E.","year":"2005","unstructured":"M. E. Taylor, P. Stone, and Y. Liu. Value functions for RL-based behavior transfer: A comparative study. In Proceedings of the Twentieth National Conference on Artificial Intelligence, July 2005. To appear."},{"key":"e_1_3_2_1_16_1","unstructured":"S. Thrun. Efficient exploration in reinforcement learning. Technical Report C I-CS-92-102 Carnegie Mellon University January 1992."},{"key":"e_1_3_2_1_17_1","volume-title":"Advances in Neural Information Processing Systems 7","author":"Thrun S.","year":"1995","unstructured":"S. Thrun and A. Schwartz. Finding structure in reinforcement learning. In Advances in Neural Information Processing Systems 7. MIT Press., 1995."},{"key":"e_1_3_2_1_18_1","unstructured":"W. T. B. Uther. Tree Based Hierarchical Reinforcement Learning. PhD thesis Carnegie Mellon University August 2002."},{"key":"e_1_3_2_1_19_1","unstructured":"C. J. C. H. Watkins. Learning from Delayed Rewards. PhD thesis King's College Cambridge UK 1989."}],"event":{"name":"AAMAS06: AAMAS '06 - 5th International Joint Conference on Autonomous Agents and Multi-agent Systems 2006","location":"Hakodate Japan","acronym":"AAMAS06","sponsor":["IFMAS The International Foundation for Multiagent Systems","SIGAI ACM Special Interest Group on Artificial Intelligence","ATAL The International Workshop on Agent Theories, Architectures, and Languages"]},"container-title":["Proceedings of the fifth international joint conference on Autonomous agents and multiagent systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1160633.1160762","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/1160633.1160762","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T17:27:06Z","timestamp":1774373226000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1160633.1160762"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2006,5,8]]},"references-count":19,"alternative-id":["10.1145\/1160633.1160762","10.1145\/1160633"],"URL":"https:\/\/doi.org\/10.1145\/1160633.1160762","relation":{},"subject":[],"published":{"date-parts":[[2006,5,8]]},"assertion":[{"value":"2006-05-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}