{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T20:02:56Z","timestamp":1772654576189,"version":"3.50.1"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,12,9]],"date-time":"2023-12-09T00:00:00Z","timestamp":1702080000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,9]],"date-time":"2023-12-09T00:00:00Z","timestamp":1702080000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"name":"China Academy of Launch Vehicle Technology","award":["CALT2022-18"],"award-info":[{"award-number":["CALT2022-18"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62306088"],"award-info":[{"award-number":["62306088"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Key Program of the National Natural Science Foundation of China","award":["51935005"],"award-info":[{"award-number":["51935005"]}]},{"name":"Basic Research Project","award":["JCKY20200603C010"],"award-info":[{"award-number":["JCKY20200603C010"]}]},{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province of China","doi-asserted-by":"crossref","award":["LH2021F023"],"award-info":[{"award-number":["LH2021F023"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Science and Technology Planning Project of Heilongjiang Province of China","award":["GA21C031"],"award-info":[{"award-number":["GA21C031"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10458-023-09630-9","type":"journal-article","created":{"date-parts":[[2023,12,9]],"date-time":"2023-12-09T06:01:41Z","timestamp":1702101701000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["IOB: integrating optimization transfer and behavior transfer for multi-policy reuse"],"prefix":"10.1007","volume":"38","author":[{"given":"Siyuan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chongjie","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,9]]},"reference":[{"issue":"3","key":"9630_CR1","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/0885-2014(91)90038-F","volume":"6","author":"SR Guberman","year":"1991","unstructured":"Guberman, S. R., & Greenfield, P. M. (1991). Learning and transfer in everyday cognition. Cognitive Development, 6(3), 233\u2013260.","journal-title":"Cognitive Development"},{"issue":"7676","key":"9630_CR2","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., Schrittwieser, J., Simonyan, K., Antonoglou, I., Huang, A., Guez, A., Hubert, T., Baker, L., Lai, M., & Bolton, A. (2017). Mastering the game of go without human knowledge. Nature, 550(7676), 354\u2013359.","journal-title":"Nature"},{"issue":"7782","key":"9630_CR3","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., Babuschkin, I., Czarnecki, W. M., Mathieu, M., Dudzik, A., Chung, J., Choi, D. H., Powell, R., Ewalds, T., & Georgiev, P. (2019). Grandmaster level in starcraft ii using multi-agent reinforcement learning. Nature, 575(7782), 350\u2013354.","journal-title":"Nature"},{"key":"9630_CR4","unstructured":"Ceron, J.S.O., & Castro, P.S. (2021). Revisiting rainbow: Promoting more insightful and inclusive deep reinforcement learning research. In: International Conference on Machine Learning, pp. 1373\u20131383 . PMLR"},{"key":"9630_CR5","doi-asserted-by":"crossref","unstructured":"Fern\u00e1ndez, F., & Veloso, M. (2006). Probabilistic policy reuse in a reinforcement learning agent. In: Proceedings of the Fifth International Joint Conference on Autonomous Agents and Multiagent Systems, pp. 720\u2013727","DOI":"10.1145\/1160633.1160762"},{"key":"9630_CR6","unstructured":"Barreto, A., Borsa, D., Quan, J., Schaul, T., Silver, D., Hessel, M., Mankowitz, D., Zidek, A., & Munos, R. (2018) Transfer in deep reinforcement learning using successor features and generalised policy improvement. In: International Conference on Machine Learning, pp. 501\u2013510. PMLR"},{"key":"9630_CR7","unstructured":"Li, S., Wang, R., Tang, M., & Zhang, C. (2019). Hierarchical reinforcement learning with advantage-based auxiliary rewards. Advances in Neural Information Processing Systems 32"},{"key":"9630_CR8","doi-asserted-by":"crossref","unstructured":"Yang, T., Hao, J., Meng, Z., Zhang, Z., Hu, Y., Chen, Y., Fan, C., Wang, W., Liu, W., Wang, Z., & Peng, J. (2020). Efficient deep reinforcement learning via adaptive policy transfer. In: Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI-20, pp. 3094\u20133100","DOI":"10.24963\/ijcai.2020\/428"},{"key":"9630_CR9","unstructured":"Zhang, J., Li, S., & Zhang, C. (2022). Cup: Critic-guided policy reuse. In: Advances in Neural Information Processing Systems"},{"key":"9630_CR10","unstructured":"Li, S., Gu, F., Zhu, G., & Zhang, C. (2018). Context-aware policy reuse. arXiv preprint arXiv:1806.03793"},{"key":"9630_CR11","unstructured":"Teh, Y., Bapst, V., Czarnecki, W.M., Quan, J., Kirkpatrick, J., Hadsell, R., Heess, N., & Pascanu, R. (2017). Distral: Robust multitask reinforcement learning. Advances in Neural Information Processing systems 30"},{"key":"9630_CR12","unstructured":"Barreto, A., Dabney, W., Munos, R., Hunt, J.J., Schaul, T., Hasselt, H.P., & Silver, D. (2017). Successor features for transfer in reinforcement learning. Advances in Neural Information Processing Systems 30"},{"key":"9630_CR13","first-page":"5587","volume":"33","author":"C-A Cheng","year":"2020","unstructured":"Cheng, C.-A., Kolobov, A., & Agarwal, A. (2020). Policy improvement via imitation of multiple oracles. Advances in Neural Information Processing Systems, 33, 5587\u20135598.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"5","key":"9630_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3453160","volume":"54","author":"S Pateria","year":"2021","unstructured":"Pateria, S., Subagdja, B., Tan, A.-H., & Quek, C. (2021). Hierarchical reinforcement learning: A comprehensive survey. ACM Computing Surveys (CSUR), 54(5), 1\u201335.","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"9630_CR15","unstructured":"Lillicrap, T.P., Hunt, J.J., Pritzel, A., Heess, N., Erez, T., Tassa, Y., Silver, D., &Wierstra, D. (2016). Continuous control with deep reinforcement learning. In: ICLR (Poster)"},{"key":"9630_CR16","unstructured":"Fujimoto, S., Hoof, H., & Meger, D. (2018). Addressing function approximation error in actor-critic methods. In: International Conference on Machine Learning, pp. 1587\u20131596. PMLR"},{"key":"9630_CR17","unstructured":"Haarnoja, T., Zhou, A., Hartikainen, K., Tucker, G., Ha, S., Tan, J., Kumar, V., Zhu, H., Gupta, A., Abbeel, P., & et al. (2018) Soft actor-critic algorithms and applications. arXiv preprint arXiv:1812.05905"},{"key":"9630_CR18","unstructured":"Yu, T., Quillen, D., He, Z., Julian, R., Hausman, K., Finn, C., & Levine, S. (2020). Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning. In: Conference on Robot Learning, pp. 1094\u20131100. PMLR"},{"key":"9630_CR19","unstructured":"Zhu, Z., Lin, K., Zhou, & J. (2020). Transfer learning in deep reinforcement learning: A survey. arXiv preprint arXiv:2009.07888"},{"key":"9630_CR20","unstructured":"Parisotto, E., Ba, J.L., & Salakhutdinov, R. (2015). Actor-mimic: Deep multitask and transfer reinforcement learning. arXiv preprint arXiv:1511.06342"},{"issue":"4","key":"9630_CR21","doi-asserted-by":"publisher","first-page":"601","DOI":"10.1109\/TEVC.2017.2664665","volume":"21","author":"Y Hou","year":"2017","unstructured":"Hou, Y., Ong, Y.-S., Feng, L., & Zurada, J. M. (2017). An evolutionary transfer reinforcement learning framework for multiagent systems. IEEE Transactions on Evolutionary Computation, 21(4), 601\u2013615.","journal-title":"IEEE Transactions on Evolutionary Computation"},{"key":"9630_CR22","doi-asserted-by":"crossref","unstructured":"Laroche, R., & Barlier, M. (2017). Transfer reinforcement learning with shared dynamics. In: Thirty-First AAAI Conference on Artificial Intelligence","DOI":"10.1609\/aaai.v31i1.10796"},{"issue":"196","key":"9630_CR23","first-page":"1","volume":"21","author":"L Lehnert","year":"2020","unstructured":"Lehnert, L., & Littman, M. L. (2020). Successor features combine elements of model-free and model-based reinforcement learning. Journal of Machine Learning Research, 21(196), 1\u201353.","journal-title":"Journal of Machine Learning Research"},{"key":"9630_CR24","doi-asserted-by":"crossref","unstructured":"Barekatain, M., Yonetani, R., & Hamaya, M. (2020). Multipolar: Multi-source policy aggregation for transfer reinforcement learning between diverse environmental dynamics. In: Proceedings of the Twenty-Ninth International Conference on International Joint Conferences on Artificial Intelligence, pp. 3108\u20133116","DOI":"10.24963\/ijcai.2020\/430"},{"key":"9630_CR25","doi-asserted-by":"crossref","unstructured":"Li, S., & Zhang, C. (2018) An optimal online method of selecting source policies for reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 32","DOI":"10.1609\/aaai.v32i1.11718"},{"key":"9630_CR26","unstructured":"Gimelfarb, M., Sanner, S., & Lee, C.-G. (2021). Contextual policy transfer in reinforcement learning domains via deep mixtures-of-experts. In: Uncertainty in Artificial Intelligence, pp. 1787\u20131797. PMLR"},{"issue":"9","key":"9630_CR27","doi-asserted-by":"publisher","first-page":"4727","DOI":"10.1109\/TNNLS.2021.3059912","volume":"33","author":"X Yang","year":"2021","unstructured":"Yang, X., Ji, Z., Wu, J., Lai, Y.-K., Wei, C., Liu, G., & Setchi, R. (2021). Hierarchical reinforcement learning with universal policies for multistep robotic manipulation. IEEE Transactions on Neural Networks and Learning Systems, 33(9), 4727\u20134741.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"9630_CR28","unstructured":"Rusu, A.A., Rabinowitz, N.C., Desjardins, G., Soyer, H., Kirkpatrick, J., Kavukcuoglu, K., Pascanu, R., & Hadsell, R. (2016). Progressive neural networks. arXiv preprint arXiv:1606.04671"},{"key":"9630_CR29","unstructured":"Berseth, G., Xie, C., Cernek, P., & Panne, M. (2018). Progressive reinforcement learning with distillation for multi-skilled motion control. arXiv preprint arXiv:1802.04765"},{"key":"9630_CR30","unstructured":"Schwarz, J., Czarnecki, W., Luketina, J., Grabska-Barwinska, A., Teh, Y.W., Pascanu, R., & Hadsell, R. (2018). Progress & compress: A scalable framework for continual learning. In: International Conference on Machine Learning, pp. 4528\u20134537. PMLR"},{"key":"9630_CR31","doi-asserted-by":"crossref","unstructured":"Mallya, A., & Lazebnik, S. (2018). Packnet: Adding multiple tasks to a single network by iterative pruning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7765\u20137773","DOI":"10.1109\/CVPR.2018.00810"},{"key":"9630_CR32","unstructured":"Duan, Y., Schulman, J., Chen, X., Bartlett, P.L., Sutskever, I., & Abbeel, P. (2016). Rl $$^2$$: Fast reinforcement learning via slow reinforcement learning. arXiv preprint arXiv:1611.02779"},{"key":"9630_CR33","unstructured":"Finn, C., Abbeel, P., & Levine, S. (2017) Model-agnostic meta-learning for fast adaptation of deep networks. In: International Conference on Machine Learning, pp. 1126\u20131135. PMLR"},{"key":"9630_CR34","unstructured":"Rakelly, K., Zhou, A., Finn, C., Levine, S., & Quillen, D. (2019). Efficient off-policy meta-reinforcement learning via probabilistic context variables. In: International Conference on Machine Learning, pp. 5331\u20135340. PMLR"},{"key":"9630_CR35","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., & Levine, S. (2018). Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: International Conference on Machine Learning, pp. 1861\u20131870. PMLR"},{"key":"9630_CR36","unstructured":"Lan, Q., Pan, Y., Fyshe, A., & White, M. (2020). Maxmin q-learning: Controlling the estimation bias of q-learning. arXiv preprint arXiv:2002.06487"},{"key":"9630_CR37","unstructured":"Kuznetsov, A., Shvechikov, P., Grishin, A., & Vetrov, D. (2020). Controlling overestimation bias with truncated mixture of continuous distributional quantile critics. In: International Conference on Machine Learning, pp. 5556\u20135566. PMLR"},{"key":"9630_CR38","unstructured":"Zhang, S., & Sutton, R.S. (2017). A deeper look at experience replay. arXiv preprint arXiv:1712.01275"},{"key":"9630_CR39","unstructured":"Fedus, W., Ramachandran, P., Agarwal, R., Bengio, Y., Larochelle, H., Rowland, M., & Dabney, W. (2020). Revisiting fundamentals of experience replay. In: International Conference on Machine Learning, pp. 3061\u20133071. PMLR"},{"key":"9630_CR40","unstructured":"Khetarpal, K., Riemer, M., Rish, I., & Precup, D. (2020). Towards continual reinforcement learning: A review and perspectives. arXiv preprint arXiv:2012.13490"},{"key":"9630_CR41","first-page":"28496","volume":"34","author":"M Wolczyk","year":"2021","unstructured":"Wolczyk, M., Zajkac, M., Pascanu, R., Kucinski, L., & Milos, P. (2021). Continual world: A robotic benchmark for continual reinforcement learning. Advances in Neural Information Processing Systems, 34, 28496\u201328510.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9630_CR42","first-page":"4767","volume":"33","author":"R Yang","year":"2020","unstructured":"Yang, R., Xu, H., Wu, Y., & Wang, X. (2020). Multi-task reinforcement learning with soft modularization. Advances in Neural Information Processing Systems, 33, 4767\u20134777.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9630_CR43","unstructured":"Sodhani, S., Zhang, A., & Pineau, J. (2021). Multi-task reinforcement learning with context-based representations. In: International Conference on Machine Learning, pp. 9767\u20139779. PMLR"},{"key":"9630_CR44","unstructured":"Wan, M., Gangwani, T., & Peng, J. (2020) Mutual information based knowledge transfer under state-action dimension mismatch. arXiv preprint arXiv:2006.07041"},{"key":"9630_CR45","unstructured":"Zhang, Q., Xiao, T., Efros, A.A., Pinto, L., & Wang, X. (2020). Learning cross-domain correspondence for control with dynamics cycle-consistency. arXiv preprint arXiv:2012.09811"},{"key":"9630_CR46","unstructured":"Heng, Y., Yang, T., ZHENG, Y., Jianye, H., & Taylor, M.E. (2022). Cross-domain adaptive transfer reinforcement learning based on state-action correspondence. In: The 38th Conference on Uncertainty in Artificial Intelligence"},{"key":"9630_CR47","first-page":"4199","volume":"33","author":"E Pol","year":"2020","unstructured":"Pol, E., Worrall, D., Hoof, H., Oliehoek, F., & Welling, M. (2020). MDP homomorphic networks: Group symmetries in reinforcement learning. Advances in Neural Information Processing Systems, 33, 4199\u20134210.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9630_CR48","unstructured":"Pol, E., Kipf, T., Oliehoek, F.A., & Welling, M. (2020). Plannable approximations to MDP homomorphisms: Equivariance under actions. arXiv preprint arXiv:2002.11963"},{"issue":"6","key":"9630_CR49","doi-asserted-by":"publisher","first-page":"1491","DOI":"10.1109\/TIT.2003.811927","volume":"49","author":"AA Fedotov","year":"2003","unstructured":"Fedotov, A. A., Harremo\u00ebs, P., & Topsoe, F. (2003). Refinements of pinsker\u2019s inequality. IEEE Transactions on Information Theory, 49(6), 1491\u20131498.","journal-title":"IEEE Transactions on Information Theory"},{"key":"9630_CR50","unstructured":"Kakade, S., & Langford, J. (2002). Approximately optimal approximate reinforcement learning. In: In Proceedings of 19th International Conference on Machine Learning. Citeseer"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09630-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-023-09630-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-023-09630-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T23:43:53Z","timestamp":1719877433000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-023-09630-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,9]]},"references-count":50,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["9630"],"URL":"https:\/\/doi.org\/10.1007\/s10458-023-09630-9","relation":{},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12,9]]},"assertion":[{"value":"7 November 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 December 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"3"}}