{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T14:29:25Z","timestamp":1773930565859,"version":"3.50.1"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s11432-023-3862-1","type":"journal-article","created":{"date-parts":[[2024,7,25]],"date-time":"2024-07-25T03:03:04Z","timestamp":1721876584000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Multi-agent policy transfer via task relationship modeling"],"prefix":"10.1007","volume":"67","author":[{"given":"Rongjun","family":"Qin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tonghan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoran","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yipeng","family":"Kang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zongzhang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chongjie","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,7,22]]},"reference":[{"key":"3862_CR1","doi-asserted-by":"publisher","first-page":"743","DOI":"10.1177\/001872678403700903","volume":"37","author":"D Tjosvold","year":"1984","unstructured":"Tjosvold D. Cooperation theory and organizations. Hum Relat, 1984, 37: 743\u2013767","journal-title":"Hum Relat"},{"key":"3862_CR2","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1007\/978-3-030-60990-0_12","volume-title":"Handbook of Reinforcement Learning and Control","author":"K Q Zhang","year":"2021","unstructured":"Zhang K Q, Yang Z R, Ba\u015far T. Multi-agent reinforcement learning: a selective overview of theories and algorithms. In: Handbook of Reinforcement Learning and Control. Berlin: Springer, 2021. 321\u2013384"},{"key":"3862_CR3","unstructured":"Lowe R, Wu Y, Tamar A, et al. Multi-agent actor-critic for mixed cooperative-competitive environments. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, 2017. 6382\u20136393"},{"key":"3862_CR4","unstructured":"Rashid T, Samvelyan M, Schroeder C, et al. QMIX: monotonic value function factorisation for deep multi-agent reinforcement learning. In: Proceedings of the 35th International Conference on Machine Learning, 2018. 4292\u20134301"},{"key":"3862_CR5","unstructured":"Wang J H, Ren Z Z, Liu T, et al. QPLEX: duplex dueling multi-agent Q-learning. In: Proceedings of the 9th International Conference on Learning Representations, 2021"},{"key":"3862_CR6","unstructured":"Wang Y H, Han B N, Wang T H, et al. DOP: off-policy multi-agent decomposed policy gradients. In: Proceedings of the 9th International Conference on Learning Representations, 2021"},{"key":"3862_CR7","unstructured":"Cao J H, Yuan L, Wang J H, et al. LINDA: multi-agent local information decomposition for awareness of teammates. 2021. ArXiv:2109.12508"},{"key":"3862_CR8","unstructured":"Foerster J, Assael Y M, de Freitas N, et al. Learning to communicate with deep multi-agent reinforcement learning. In: Proceedings of the 30th International Conference on Neural Information Processing Systems, 2016. 2145\u20132153"},{"key":"3862_CR9","unstructured":"Jiang J C, Lu Z Q. Learning attentional communication for multi-agent cooperation. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems, 2018. 7265\u20137275"},{"key":"3862_CR10","unstructured":"Jiang J C, Dun C, Huang T J, et al. Graph convolutional reinforcement learning. In: Proceedings of the 8th International Conference on Learning Representations, 2020"},{"key":"3862_CR11","unstructured":"Wang T H, Dong H, Lesser V, et al. ROMA: multi-agent reinforcement learning with emergent roles. In: Proceedings of the 37th International Conference on Machine Learning, 2020. 9876\u20139886"},{"key":"3862_CR12","unstructured":"Wang T H, Gupta T, Mahajan A, et al. RODE: learning roles to decompose multi-agent tasks. In: Proceedings of the 9th International Conference on Learning Representations, 2021"},{"key":"3862_CR13","unstructured":"Zhu Z D, Lin K X, Zhou J Y, et al. Transfer learning in deep reinforcement learning: a survey. 2020. ArXiv:2009.07888"},{"key":"3862_CR14","unstructured":"Long Q, Zhou Z H, Gupta A, et al. Evolutionary population curriculum for scaling multi-agent reinforcement learning. In: Proceedings of International Conference on Learning Representations, 2019"},{"key":"3862_CR15","doi-asserted-by":"crossref","unstructured":"Wang W X, Yang T P, Liu Y, et al. From few to more: large-scale dynamic multiagent curriculum learning. In: Proceedings of AAAI Conference on Artificial Intelligence, 2020. 7293\u20137300","DOI":"10.1609\/aaai.v34i05.6221"},{"key":"3862_CR16","unstructured":"Agarwal A, Kumar S, Sycara K, et al. Learning transferable cooperative behavior in multi-agent teams. In: Proceedings of the 19th International Conference on Autonomous Agents and Multiagent Systems, 2020. 1741\u20131743"},{"key":"3862_CR17","unstructured":"Hu S Y, Zhu F D, Chang X J, et al. UPDeT: universal multi-agent reinforcement learning via policy decoupling with transformers. In: Proceedings of the 9th International Conference on Learning Representations, 2021"},{"key":"3862_CR18","unstructured":"Zhou T Z, Zhang F B, Shao K, et al. Cooperative multi-agent transfer learning with level-adaptive credit assignment. 2021. ArXiv:2106.00517"},{"key":"3862_CR19","unstructured":"Samvelyan M, Rashid T, de Witt C S, et al. The StarCraft multi-agent challenge. In: Proceedings of the 18th International Conference on Autonomous Agents and Multiagent Systems, 2019. 2186\u20132188"},{"key":"3862_CR20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8","volume-title":"A Concise Introduction to Decentralized POMDPs","author":"F A Oliehoek","year":"2016","unstructured":"Oliehoek F A, Amato C. A Concise Introduction to Decentralized POMDPs. Berlin: Springer, 2016"},{"key":"3862_CR21","doi-asserted-by":"publisher","first-page":"297","DOI":"10.1016\/0024-3795(94)90493-6","volume":"197","author":"\u00c5 Bj\u00f6rck","year":"1994","unstructured":"Bj\u00f6rck \u00c5. Numerics of Gram-Schmidt orthogonalization. Linear Algebra Appl, 1994, 197: 297\u2013316","journal-title":"Linear Algebra Appl"},{"key":"3862_CR22","unstructured":"Iqbal S, de Witt C S, Peng B, et al. Randomized entity-wise factorization for multi-agent reinforcement learning. In: Proceedings of the 38th International Conference on Machine Learning, 2021. 4596\u20134606"},{"key":"3862_CR23","unstructured":"Wang W X, Yang T P, Liu Y, et al. Action semantics network: considering the effects of actions in multiagent systems. In: Proceedings of the 8th International Conference on Learning Representations, 2020"},{"key":"3862_CR24","unstructured":"Barreto A, Borsa D, Quan J, et al. Transfer in deep reinforcement learning using successor features and generalised policy improvement. In: Proceedings of the 35th International Conference on Machine Learning, 2018. 501\u2013510"},{"key":"3862_CR25","unstructured":"Petangoda J C, Adam V, Vrancx P, et al. Disentangled skill embeddings for reinforcement learning. 2019. ArXiv:1906.09223"},{"key":"3862_CR26","unstructured":"Sun Y C, Zheng R J, Wang X Y, et al. Transfer RL across observation feature spaces via model-based regularization. In: Proceedings of the 10th International Conference on Learning Representations, 2022"},{"key":"3862_CR27","unstructured":"Tao Y Z, Genc S, Chung J, et al. REPAINT: knowledge transfer in deep reinforcement learning. In: Proceedings of the 38th International Conference on Machine Learning, 2021. 10141\u201310152"},{"key":"3862_CR28","unstructured":"Silva F L, Costa A H R. Transfer learning for multiagent reinforcement learning systems. In: Proceedings of the 25th International Joint Conference on Artificial Intelligence, 2016. 3982\u20133983"},{"key":"3862_CR29","doi-asserted-by":"crossref","unstructured":"Wadhwania S, Kim D K, Omidshafiei S, et al. Policy distillation and value matching in multiagent reinforcement learning. In: Proceedings of IEEE\/RSJ International Conference on Intelligent Robots and Systems, 2019. 8193\u20138200","DOI":"10.1109\/IROS40897.2019.8967849"},{"key":"3862_CR30","doi-asserted-by":"crossref","unstructured":"Omidshafiei S, Kim D K, Liu M, et al. Learning to teach in cooperative multiagent reinforcement learning. In: Proceedings of the 33rd AAAI Conference on Artificial Intelligence, 2019. 6128\u20136136","DOI":"10.1609\/aaai.v33i01.33016128"},{"key":"3862_CR31","unstructured":"Yang T P, Wang W X, Tang H Y, et al. An efficient transfer learning framework for multiagent reinforcement learning. In: Proceedings of the 35th Conference on Neural Information Processing Systems, 2021. 17037\u201317048"},{"key":"3862_CR32","unstructured":"de Hauwere Y M, Vrancx P, Now\u00e9 A. Learning multi-agent state space representations. In: Proceedings of the 9th International Conference on Autonomous Agents and Multiagent Systems, 2010. 715\u2013722"},{"key":"3862_CR33","unstructured":"Grover A, Gupta J, Burda Y, et al. Learning policy representations in multiagent systems. In: Proceedings of the 35th International Conference on Machine Learning, 2018. 1802\u20131811"},{"key":"3862_CR34","unstructured":"Xie A, Losey D, Tolsma R, et al. Learning latent representations to influence multi-agent interaction. In: Proceedings of Conference on Robot Learning, 2021. 575\u2013588"},{"key":"3862_CR35","unstructured":"Zhang S N, Shen L, Han L. Learning meta representations for agents in multi-agent reinforcement learning. 2021. ArXiv:2108.12988"},{"key":"3862_CR36","first-page":"1334","volume":"17","author":"S Levine","year":"2016","unstructured":"Levine S, Finn C, Darrell T, et al. End-to-end training of deep visuomotor policies. J Mach Learn Res, 2016, 17: 1334\u20131773","journal-title":"J Mach Learn Res"},{"key":"3862_CR37","unstructured":"Ghosh D, Singh A, Rajeswaran A, et al. Divide-and-conquer reinforcement learning. In: Proceedings of the 6th International Conference on Learning Representations, 2018"},{"key":"3862_CR38","unstructured":"Xu Z Y, Wu K, Che Z P, et al. Knowledge transfer in multi-task deep reinforcement learning for continuous control. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, 2020. 15146\u201315155"},{"key":"3862_CR39","doi-asserted-by":"crossref","unstructured":"Deisenroth M P, Englert P, Peters J, et al. Multi-task policy search for robotics. In: Proceedings of IEEE International Conference on Robotics and Automation, 2014. 3876\u20133881","DOI":"10.1109\/ICRA.2014.6907421"},{"key":"3862_CR40","unstructured":"da Silva B C, Konidaris G, Barto A G. Learning parameterized skills. In: Proceedings of the 29th International Conference on Machine Learning, 2012. 1443\u20131450"},{"key":"3862_CR41","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1007\/s10514-012-9290-3","volume":"33","author":"J Kober","year":"2012","unstructured":"Kober J, Wilhelm A, \u00d6ztop E, et al. Reinforcement learning to adjust parametrized motor primitives to new situations. Auton Robot, 2012, 33: 361\u2013379","journal-title":"Auton Robot"},{"key":"3862_CR42","unstructured":"Yang R H, Xu H Z, Wu Y, et al. Multi-task reinforcement learning with soft modularization. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, 2020. 4767\u20134777"},{"key":"3862_CR43","unstructured":"Du Y S, Czarnecki W M, Jayakumar S M, et al. Adapting auxiliary losses using gradient similarity. 2018. ArXiv:1812.02224"},{"key":"3862_CR44","unstructured":"Suteu M, Guo Y K. Regularizing deep multi-task networks using orthogonal gradients. 2019. ArXiv:1912.06844"},{"key":"3862_CR45","unstructured":"Yu T H, Kumar S, Gupta A, et al. Gradient surgery for multi-task learning. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, 2020. 5824\u20135836"},{"key":"3862_CR46","doi-asserted-by":"crossref","unstructured":"You Q, Wu O, Luo G, et al. Metadata-based clustered multi-task learning for thread mining in web communities. In: Proceedings of International Conference on Machine Learning and Data Mining in Pattern Recognition, 2016. 421\u2013434","DOI":"10.1007\/978-3-319-41920-6_33"},{"key":"3862_CR47","doi-asserted-by":"crossref","unstructured":"Zheng Z M, Wang Y Q, Dai Q Y, et al. Metadata-driven task relation discovery for multi-task learning. In: Proceedings of the 28th International Joint Conference on Artificial Intelligence, 2019. 4426\u20134432","DOI":"10.24963\/ijcai.2019\/615"},{"key":"3862_CR48","unstructured":"Sodhani S, Zhang A, Pineau J. Multi-task reinforcement learning with context-based representations. In: Proceedings of the 38th International Conference on Machine Learning, 2021. 9767\u20139779"},{"key":"3862_CR49","first-page":"13198","volume":"22","author":"L Zintgraf","year":"2021","unstructured":"Zintgraf L, Schulze S, Lu C, et al. VariBAD: variational Bayes-adaptive deep RL via meta-learning. J Mach Learn Res, 2021, 22: 13198\u201313236","journal-title":"J Mach Learn Res"},{"key":"3862_CR50","unstructured":"Rakelly K, Zhou A, Finn C, et al. Efficient off-policy meta-reinforcement learning via probabilistic context variables. In: Proceedings of the 36th International Conference on Machine Learning, 2019. 5331\u20135340"},{"key":"3862_CR51","unstructured":"Wang J X, Kurth-Nelson Z, Soyer H, et al. Learning to reinforcement learn. In: Proceedings of Annual Meeting on Cognitive Science Society, 2017"},{"key":"3862_CR52","unstructured":"Duan Y, Schulman J, Chen X, et al. RL2: fast reinforcement learning via slow reinforcement learning. 2016. ArXiv:1611.02779"},{"key":"3862_CR53","unstructured":"Hausman K, Springenberg J T, Wang Z Y, et al. Learning an embedding space for transferable robot skills. In: Proceedings of the 6th International Conference on Learning Representations, 2018"},{"key":"3862_CR54","doi-asserted-by":"crossref","unstructured":"Arnekvist I, Kragic D, Stork J A. VPE: variational policy embedding for transfer reinforcement learning. In: Proceedings of International Conference on Robotics and Automation, 2019. 36\u201342","DOI":"10.1109\/ICRA.2019.8793556"},{"key":"3862_CR55","unstructured":"Co-Reyes J D, Liu Y X, Gupta A, et al. Self-consistent trajectory autoencoder: hierarchical reinforcement learning with trajectory embeddings. In: Proceedings of the 35th International Conference on Machine Learning, 2018. 1637\u20131647"},{"key":"3862_CR56","unstructured":"Zhang A, Satija H, Pineau J. Decoupling dynamics and reward for transfer learning. In: Proceedings of the 6th International Conference on Learning Representations, 2018"},{"key":"3862_CR57","doi-asserted-by":"crossref","unstructured":"Lan L, Li Z G, Guan X H, et al. Meta reinforcement learning with task embedding and shared policy. In: Proceedings of the 28th International Joint Conference on Artificial Intelligence, 2019. 2794\u20132800","DOI":"10.24963\/ijcai.2019\/387"},{"key":"3862_CR58","unstructured":"Wang T W, Liao R J, Ba J, et al. NerveNet: learning structured policy with graph neural networks. In: Proceedings of the 6th International Conference on Learning Representations, 2018"},{"key":"3862_CR59","unstructured":"Pathak D, Lu C, Darrell T, et al. Learning to control self-assembling morphologies: a study of generalization via modularity. In: Proceedings of the 33rd International Conference on Neural Information Processing Systems, 2019. 2295\u20132305"},{"key":"3862_CR60","unstructured":"Huang W L, Mordatch I, Pathak D. One policy to control them all: shared modular policies for agent-agnostic control. In: Proceedings of the 37th International Conference on Machine Learning, 2020. 4455\u20134464"},{"key":"3862_CR61","unstructured":"Kurin V, Igl M, Rockt\u00e4schel T, et al. My body is a cage: the role of morphology in graph-based incompatible control. In: Proceedings of the 9th International Conference on Learning Representations, 2021"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-023-3862-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-023-3862-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-023-3862-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T19:40:11Z","timestamp":1758310811000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-023-3862-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,22]]},"references-count":61,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2024,8]]}},"alternative-id":["3862"],"URL":"https:\/\/doi.org\/10.1007\/s11432-023-3862-1","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,22]]},"assertion":[{"value":"27 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 June 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"182101"}}