{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T03:10:08Z","timestamp":1777605008177,"version":"3.51.4"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2012,10,13]],"date-time":"2012-10-13T00:00:00Z","timestamp":1350086400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Prog Artif Intell"],"published-print":{"date-parts":[[2013,3]]},"DOI":"10.1007\/s13748-012-0026-6","type":"journal-article","created":{"date-parts":[[2012,10,17]],"date-time":"2012-10-17T12:08:21Z","timestamp":1350475701000},"page":"13-27","source":"Crossref","is-referenced-by-count":24,"title":["Learning domain structure through probabilistic policy reuse in reinforcement learning"],"prefix":"10.1007","volume":"2","author":[{"given":"Fernando","family":"Fern\u00e1ndez","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manuela","family":"Veloso","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,10,13]]},"reference":[{"key":"26_CR1","first-page":"237","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling, L.P., Littman, M.L., Moore, A.W.: Reinforcement learning: a survey. Int. J. Artif. Intell. Res. 4, 237 (1996)","journal-title":"Int. J. Artif. Intell. Res."},{"key":"26_CR2","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"key":"26_CR3","unstructured":"Watkins, C.: Learning from delayed rewards. Ph.D. thesis, Cambridge University, Cambridge (1989)"},{"key":"26_CR4","first-page":"257","volume":"8","author":"G Tesauro","year":"1992","unstructured":"Tesauro, G.: Practical issues in temporal difference learning. Mach. Learn. 8, 257 (1992)","journal-title":"Mach. Learn."},{"key":"26_CR5","doi-asserted-by":"crossref","unstructured":"Stone, P., Sutton, R.S., Kuhlmann, G.: Reinforcement learning for RoboCup-soccer keepaway. Adapt. Behav. 13(3) (2005)","DOI":"10.1177\/105971230501300301"},{"key":"26_CR6","unstructured":"Taylor, M.E., Stone, P., Liu, Y.: Inter-task action correlation for reinforcement learning tasks. In: Proceedings of the Twentieth National Conference on Artificial Intelligence (AAAI\u201905) (2005)"},{"key":"26_CR7","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton, R.S., Precup, D., Singh, S.: Between mdps and semi-mdps: a framework for temporal abstraction in reinforcement learning. Artif. Intell. 112, 181 (1999)","journal-title":"Artif. Intell."},{"key":"26_CR8","first-page":"2259","volume":"7","author":"A Jonsson","year":"2006","unstructured":"Jonsson, A., Barto, A.: Causal graph based decomposition of factored mdps. J. Mach. Learn. Res. 7, 2259 (2006)","journal-title":"J. Mach. Learn. Res."},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"Veloso, M.M.: Planning and Learning by Analogical Reasoning. Springer, Berlin (1994) (Revised PhD Thesis Manuscript, Carnegie Mellon University, technical report CMU-CS-92-174)","DOI":"10.1007\/3-540-58811-6"},{"key":"26_CR10","doi-asserted-by":"crossref","unstructured":"Bruce, J., Veloso, M.: Real-time randomized path planning for robot navigation. In: Proceedings of IROS-2002 Switzerland (2002). (An earlier version of this paper appears in the Proceedings of the RoboCup-2002 Symposium)","DOI":"10.1109\/IRDS.2002.1041624"},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"Taylor, M., Stone, P.: An introduction to intertask transfer for reinforcement learning. AI Magazine 32(1), (2012)","DOI":"10.1609\/aimag.v32i1.2329"},{"key":"26_CR12","unstructured":"Fern\u00e1ndez, F., Veloso, M.: Policy reuse for transfer learning across tasks with different state and action spaces. In: ICML\u201906 Workshop on Structural Knowledge Transfer for, Machine Learning (2006)"},{"key":"26_CR13","unstructured":"Garc\u00eda, F.J., Veloso, M., Fern\u00e1ndez, F.: Reinforcement learning in the robocup-soccer keepaway. In: Proceedings of the 12th Conference of the Spanish Association for, Artificial Intelligence (CAEPIA\u201907+TTIA) (2007)"},{"issue":"7","key":"26_CR14","doi-asserted-by":"crossref","first-page":"866","DOI":"10.1016\/j.robot.2010.03.007","volume":"58","author":"F Fern\u00e1ndez","year":"2010","unstructured":"Fern\u00e1ndez, F., Garc\u00eda, J., Veloso, M.: Probabilistic policy reuse for inter-task transfer learning. Robot. Autonom. Syst. 58(7), 866 (2010). doi: 10.1016\/j.robot.2010.03.007","journal-title":"Robot. Autonom. Syst."},{"key":"26_CR15","doi-asserted-by":"crossref","unstructured":"Dasgupta, P., Cheng, K., Banerjee, B.: Adaptive multi-robot team reconfiguration using a policy-reuse reinforcement learning approach. Advanced agent technology. In: Dechesne, F., Hattori, H., Mors, A., Such, J., Weyns, D., Dignum, F. (eds.) Lecture Notes in Computer Science, vol. 7068, pp. 330\u2013345. Springer, Berlin (2012)","DOI":"10.1007\/978-3-642-27216-5_23"},{"key":"26_CR16","unstructured":"Taylor, M.E., Suay, H.B., Chernova, S.: Integrating reinforcement learning with human demonstrations of varying ability. In: The 10th International Conference on Autonomous Agents and Multiagent Systems, vol. 2, AAMAS\u201911, pp. 617\u2013624. International Foundation for Autonomous Agents and Multiagent Systems, Richland (2011). http:\/\/dl.acm.org\/citation.cfm?id=2031678.2031705"},{"key":"26_CR17","unstructured":"da Silva, B.N., Mackworth, A.: Using spatial hints to improve policy reuse in a reinforcement learning agent. In: Proceedings of the Autonomous Agents and Multi agent Systems, pp. 317\u2013324 (AAMAS 2010) (2010)"},{"key":"26_CR18","unstructured":"Thrun, S.: Efficient exploration in reinforcement learning. Tech. Rep. C, I-CS-92-102, Carnegie Mellon University (1992)"},{"key":"26_CR19","unstructured":"Maclin, R., Shavlik, J., Torrey, L., Walker, T., Wild, E.: Giving advice about preferred actions to reinforcement learners via knowledge-based kernel regression. In: Proceedings of the Twentieth National Conference on Artificial Intelligence (2005)"},{"key":"26_CR20","unstructured":"Smart, W.D., Kaelbling, L.P.: Practical reinforcement learning in continuous spaces. In: Proceedings of the International Conference of, Machine Learning, pp. 903\u2013907 (2000)"},{"key":"26_CR21","doi-asserted-by":"crossref","first-page":"569","DOI":"10.1613\/jair.898","volume":"19","author":"B Price","year":"2003","unstructured":"Price, B., Boutilier, C.: Accelerating reinforcement learning through implicit imitation. J. Artif. Intell. Res. 19, 569 (2003)","journal-title":"J. Artif. Intell. Res."},{"key":"26_CR22","unstructured":"Carroll, J., Peterson, T., Owens, N.: Memory-guided exploration in reinforcement learning. In: Proceedings of the Internatioanal Joint Conference on, Neural Networks (2001)"},{"key":"26_CR23","unstructured":"Dixon, K., Malak, R., Khos, P.: Incorporating prior knowledge and previously learned information into reinforcement learning agents. Carnegie Mellon University, Institute for Complex Engineered Systems, Tech. rep. (2000)"},{"key":"26_CR24","unstructured":"Carroll, J., Peterson, T.: Fixed vs. dynamic sub-transfer in reinforcement learning. In: Proceedings of the International Conference on Machine Learning and Applications (2002)"},{"key":"26_CR25","doi-asserted-by":"crossref","first-page":"375","DOI":"10.1023\/B:AIRE.0000036264.95672.64","volume":"21","author":"MG Madden","year":"2004","unstructured":"Madden, M.G., Howley, T.: Transfer of experience between reinforcement learning environments with progressive difficulty. Artif. Intell. Rev. 21, 375 (2004)","journal-title":"Artif. Intell. Rev."},{"issue":"1","key":"26_CR26","first-page":"2125","volume":"8","author":"M Taylor","year":"2007","unstructured":"Taylor, M., Stone, P., Liu, Y.: Transfer learning via inter-task mappings for temporal difference learning. J. Mach. Learn. Res. 8(1), 2125 (2007)","journal-title":"J. Mach. Learn. Res."},{"key":"26_CR27","unstructured":"Taylor, M.E., Stone, P.: Value functions for RL-based behavior transfer: A comparative study. In: Proceedings of the Twenty-first National Conference on, Artificial Intelligence (AAAI\u201906) (2006)"},{"key":"26_CR28","unstructured":"Walsh, T.J., Li, L., Littman, M.: Transferring state abstractions between mdps. In: Proceedings of the ICML\u2019 06 Workshop on Structural Knowledge Tranfer for, Machine Learning (2006)"},{"key":"26_CR29","unstructured":"Soni, V., Singh, S.: Using homomorphisms to transfer options across continuous reinforcement learning domains. In: Proceedings of the National Conference on Artificial Intelligence (AAAI\u201906) (2006)"},{"key":"26_CR30","unstructured":"Uther, W.T.B.: Tree based hierarchical reinforcement learning. Ph.D. thesis, Carnegie Mellon University (2002)"},{"key":"26_CR31","unstructured":"Torrey, L., Shavlik, J., Walker, T., Maclin, R.: Relational macros for transfer in reinforcement learning. In: Proceedings of 17th Conference on Inductive Logic Programming (2007)"},{"key":"26_CR32","unstructured":"Sutton, R.S., Precup, D., Singh, S.: Intra-option learning about temporally abstract actions. In: Proceedings of the Internacional Conference on, Machine Learning (ICML\u201998) (1998)"},{"key":"26_CR33","doi-asserted-by":"crossref","unstructured":"Stolle, M., Precup, D.: Learning options in reinforcement learning. In: Proceedings of the 5th International Symposium on Abstraction, Reformulation and Approximation, Lecture Notes In Computer Science, vol. 2371. Springer, Berlin (2002)","DOI":"10.1007\/3-540-45622-8_16"},{"key":"26_CR34","doi-asserted-by":"crossref","unstructured":"Taylor, M., Stone, P.: Cross-domain transfer for reinforcement learning. In: Proceedings of the 24th International Conference on, Machine Learning (ICML\u201907) (2007)","DOI":"10.1145\/1273496.1273607"},{"key":"26_CR35","doi-asserted-by":"crossref","unstructured":"Singh, S.P.: Transfer of learning by composing solutions of elemental sequential tasks. Mach. Learn. 8 (1992)","DOI":"10.1007\/BF00992700"},{"key":"26_CR36","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"TG Dietterich","year":"2000","unstructured":"Dietterich, T.G.: Hierarchical reinforcement learning with the MAXQ value function decomposition. J. Artif. Intell. Res. 13, 227 (2000)","journal-title":"J. Artif. Intell. Res."},{"key":"26_CR37","unstructured":"Hengst, B.: Discovering hierarchy in reinforcement learning with HEXQ. In: Proceedings of the Nineteenth International Conference on, Machine Learning (2002)"},{"key":"26_CR38","unstructured":"Thrun, S., Schwartz, A.: Finding structure in reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a07. MIT Press, Massachusetts (1995)"},{"key":"26_CR39","doi-asserted-by":"crossref","unstructured":"Simsek, O., Wolfe, A.P., Barto, A.G.: Identifying useful subgoals in reinforcement learning by local graph partitioning. In: Proceedings of the Twenty-Second International Conference on Machine Learning (2005)","DOI":"10.1145\/1102351.1102454"},{"key":"26_CR40","unstructured":"Bowling, M., Veloso, M.: Bounding the suboptimality of reusing subproblems. In: Proceedings of IJCAI-99 (1999)"},{"key":"26_CR41","unstructured":"Parr, R.: Flexible decomposition algorithms for weakly coupled markov decision problems. In: Proceedings of the 14th Annual Conference on Uncertainty in Artificial Intelligence (UAI-98). Morgan Kaufmann, San Francisco (1998)"},{"key":"26_CR42","first-page":"1633","volume":"10","author":"M Taylor","year":"2009","unstructured":"Taylor, M., Stone, P.: Transfer learning for reinforcement learning domains: A survey. J. Mach. Learn. Res. 10, 1633 (2009)","journal-title":"J. Mach. Learn. Res."},{"key":"26_CR43","unstructured":"Sherstov, A.A., Stone, P.: Improving action selection in MDP\u2019s via knowledge transfer. In: Proceedings of the Twentieth National Conference on, Artificial Intelligence (2005)"},{"key":"26_CR44","unstructured":"Fern\u00e1ndez, F., Veloso, M.: Reusing and building policy libraries. In: Proceedings of the International Conference on Automated Planning and Schedulling (ICAPS\u201906) (2006)"},{"key":"26_CR45","doi-asserted-by":"crossref","unstructured":"Yu, J.Y., Mannor, S.: Piecewise-stationary bandit problems with side observations. In: ICML\u201909: Proceedings of the 26th Annual International Conference on, Machine Learning (2009)","DOI":"10.1145\/1553374.1553524"},{"key":"26_CR46","doi-asserted-by":"crossref","first-page":"1037","DOI":"10.1002\/int.20105","volume":"20","author":"RB Ollington","year":"2005","unstructured":"Ollington, R.B., Vamplew, P.W.: Reinforcement learning for dynamic goals and environments. Int. J. Intell. Syst. 20, 1037 (2005)","journal-title":"Int. J. Intell. Syst."},{"issue":"2","key":"26_CR47","doi-asserted-by":"crossref","first-page":"213","DOI":"10.1002\/int.20255","volume":"23","author":"F Fern\u00e1ndez","year":"2008","unstructured":"Fern\u00e1ndez, F., Borrajo, D.: Two steps reinforcement learning. Int. J. Intell. Syst. 23(2), 213 (2008)","journal-title":"Int. J. Intell. Syst."},{"key":"26_CR48","doi-asserted-by":"crossref","unstructured":"Chevaleyre, Y., Pamponet, A.M., Zucker, J.D.: Experiments with adaptive transfer rate in reinforcement learning. Knowledge Acquisition: Approaches, Algorithms and Applications (PKAW\u2019 2008). In: Lecture Notes in Artificial Intelligence, vol. 5465 (2009)","DOI":"10.1007\/978-3-642-01715-5_1"}],"container-title":["Progress in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13748-012-0026-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s13748-012-0026-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13748-012-0026-6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,7,4]],"date-time":"2019-07-04T15:14:29Z","timestamp":1562253269000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s13748-012-0026-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,10,13]]},"references-count":48,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2013,3]]}},"alternative-id":["26"],"URL":"https:\/\/doi.org\/10.1007\/s13748-012-0026-6","relation":{},"ISSN":["2192-6352","2192-6360"],"issn-type":[{"value":"2192-6352","type":"print"},{"value":"2192-6360","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,10,13]]}}}