{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T12:55:38Z","timestamp":1761396938847,"version":"build-2065373602"},"reference-count":62,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2022ZD0116600"],"award-info":[{"award-number":["2022ZD0116600"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Science Foundation of China","doi-asserted-by":"publisher","award":["62276124"],"award-info":[{"award-number":["62276124"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jiangsu Science Foundation","award":["BK20243039","BK2024119"],"award-info":[{"award-number":["BK20243039","BK2024119"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["14380020"],"award-info":[{"award-number":["14380020"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Evol. Computat."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1109\/tevc.2024.3485177","type":"journal-article","created":{"date-parts":[[2024,10,23]],"date-time":"2024-10-23T13:48:13Z","timestamp":1729691293000},"page":"2229-2243","source":"Crossref","is-referenced-by-count":4,"title":["Heterogeneous Multiagent Zero-Shot Coordination by Coevolution"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6789-2670","authenticated-orcid":false,"given":"Ke","family":"Xue","sequence":"first","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6268-1268","authenticated-orcid":false,"given":"Yutong","family":"Wang","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5198-9141","authenticated-orcid":false,"given":"Cong","family":"Guan","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7803-0766","authenticated-orcid":false,"given":"Lei","family":"Yuan","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0730-8107","authenticated-orcid":false,"given":"Haobo","family":"Fu","sequence":"additional","affiliation":[{"name":"Game AI Center, Tencent AI Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Fu","sequence":"additional","affiliation":[{"name":"Game AI Center, Tencent AI Lab, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6011-2512","authenticated-orcid":false,"given":"Chao","family":"Qian","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1732-9545","authenticated-orcid":false,"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology and the School of Artificial Intelligence, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1093\/oso\/9780195099713.001.0001"},{"key":"ref2","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019","journal-title":"arXiv:1912.06680"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3071178.3071186"},{"key":"ref4","first-page":"5175","article-title":"On the utility of learning about humans for human-AI coordination","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Carroll"},{"key":"ref5","first-page":"1","article-title":"Generating diverse cooperative agents by learning incompatible policies","volume-title":"Proc. 11th Int. Conf. Learn. Represent. (ICLR)","author":"Charakorn"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-66515-9_4"},{"key":"ref7","first-page":"5032","article-title":"Improving exploration in evolution strategies for deep reinforcement learning via a population of novelty-seeking agents","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Conti"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/nature14422"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2017.2704781"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.5555\/1248547.1248548"},{"key":"ref11","first-page":"1","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. 6th Int. Conf. Learn. Represent. (ICLR)","author":"Eysenbach"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i7.16740"},{"key":"ref13","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","author":"Haarnoja"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-023-3909-x"},{"key":"ref15","first-page":"4369","article-title":"Off-belief learning","volume-title":"Proc. 38th Int. Conf. Mach. Learn. (ICML)","author":"Hu"},{"key":"ref16","first-page":"4399","article-title":"\u2018Other-play\u2019 for zero-shot coordination","volume-title":"Proc. 37th Int. Conf. Mach. Learn. (ICML)","author":"Hu"},{"key":"ref17","article-title":"Heterogeneous beliefs and multi-population learning in network games","author":"Hu","year":"2023","journal-title":"arXiv:2301.04929"},{"key":"ref18","article-title":"Population based training of neural networks","author":"Jaderberg","year":"2017","journal-title":"arXiv:1711.09846"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2023.3292527"},{"key":"ref20","first-page":"1258","article-title":"Reinforcement learning of coordination in heterogeneous cooperative multi-agent systems","volume-title":"Proc. 3rd Int. Joint Conf. Auton. Agents Multiagent Syst. (AAMAS)","author":"Kapetanakis"},{"key":"ref21","article-title":"Heterogeneous-agent mirror learning: A continuum of solutions to cooperative MARL","author":"Kuba","year":"2022","journal-title":"arXiv:2208.01682"},{"key":"ref22","first-page":"8198","article-title":"One solution is not all you need: Few-shot extrapolation via structured MaxEnt RL","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Kumar"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1162\/EVCO_a_00025"},{"key":"ref24","article-title":"Semantically aligned task decomposition in multi-agent reinforcement learning","author":"Li","year":"2023","journal-title":"arXiv:2305.10865"},{"key":"ref25","first-page":"1","article-title":"Cooperative open-ended learning framework for zero-shot coordination","volume-title":"Proc. 40th Int. Conf. Mach. Learn. (ICML)","author":"Li"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-023-3864-6"},{"key":"ref27","first-page":"941","article-title":"Towards unifying behavioral and response diversity for open-ended learning in zero-sum games","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Liu"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/336"},{"key":"ref29","first-page":"679","article-title":"PECAN: Leveraging policy ensemble for context-aware zero-shot human-AI coordination","volume-title":"Proc. 22nd Int. Conf. Auton. Agents Multiagent Syst. (AAMAS)","author":"Lou"},{"key":"ref30","first-page":"853","article-title":"Any-play: An intrinsic augmentation for zero-shot coordination","volume-title":"Proc. 21st Int. Conf. Auton. Agents Multiagent Syst. (AAMAS)","author":"Lucas"},{"key":"ref31","first-page":"7204","article-title":"Trajectory diversity for zero-shot coordination","volume-title":"Proc. 38th Int. Conf. Mach. Learn. (ICML)","author":"Lupu"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2018.2868770"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20614-6_16"},{"issue":"129","key":"ref34","first-page":"1","article-title":"On the approximation of cooperative heterogeneous multi-agent reinforcement learning (MARL) using mean field control (MFC)","volume":"23","author":"Mondal","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8"},{"key":"ref36","article-title":"Evolving curricula with regret-based environment design","author":"Parker-Holder","year":"2022","journal-title":"arXiv:2203.01302"},{"key":"ref37","first-page":"753","article-title":"Ridge rider: Finding diverse solutions by following eigenvectors of the Hessian","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Parker-Holder"},{"key":"ref38","first-page":"18050","article-title":"Effective diversity in population based reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Parker-Holder"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/671"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1162\/106365600568086"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3712255.3734227"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3148438"},{"key":"ref43","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref45","article-title":"Mastering chess and shogi by self-play with a general reinforcement learning algorithm","author":"Silver","year":"2017","journal-title":"arXiv:1712.01815"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v24i1.7529"},{"key":"ref47","article-title":"Open-ended learning leads to generally capable agents","author":"Stooke","year":"2021","journal-title":"arXiv:2107.12808"},{"key":"ref48","first-page":"14502","article-title":"Collaborating with humans without human data","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","author":"Strouse"},{"volume-title":"Reinforcement Learning: An Introduction","year":"2018","author":"Sutton","key":"ref49"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1994.6.2.215"},{"issue":"11","key":"ref51","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."},{"key":"ref52","article-title":"Starcraft II: A new challenge for reinforcement learning","author":"Vinyals","year":"2017","journal-title":"arXiv:1708.04782"},{"key":"ref53","first-page":"1279","article-title":"Co-GAIL: Learning diverse strategies for human-robot collaboration","volume-title":"Proc. 6th Conf. Robot Learn. (CoRL)","author":"Wang"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3321707.3321799"},{"key":"ref55","first-page":"9940","article-title":"Enhanced POET: Open-ended reinforcement learning through unbounded invention of learning challenges and their solutions","volume-title":"Proc. 37th Int. Conf. Mach. Learn. (ICML)","author":"Wang"},{"key":"ref56","first-page":"1","article-title":"Evolutionary diversity optimization with clustering-based selection for reinforcement learning","volume-title":"Proc. 10th Int. Conf. Learn. Represent. (ICLR)","author":"Wang"},{"key":"ref57","first-page":"1418","article-title":"Mis-spoke or mis-lead: Achieving robustness in multi-agent communicative reinforcement learning","volume-title":"Proc. 21st Int. Conf. Auton. Agents Multiagent Syst.","author":"Xue"},{"key":"ref58","first-page":"1","article-title":"Learning zero-shot cooperation with humans, assuming humans are biased","volume-title":"Proc. 11th Int. Conf. Learn. Represent. (ICLR)","author":"Yu"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-023-2733-5"},{"key":"ref60","article-title":"A survey of progress on cooperative multi-agent reinforcement learning in open environment","author":"Yuan","year":"2023","journal-title":"arXiv:2312.01058"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i10.26388"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i5.25758"}],"container-title":["IEEE Transactions on Evolutionary Computation"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/4235\/11199991\/10730789.pdf?arnumber=10730789","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T17:41:08Z","timestamp":1760377268000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10730789\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10]]},"references-count":62,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tevc.2024.3485177","relation":{},"ISSN":["1089-778X","1089-778X","1941-0026"],"issn-type":[{"type":"print","value":"1089-778X"},{"type":"print","value":"1089-778X"},{"type":"electronic","value":"1941-0026"}],"subject":[],"published":{"date-parts":[[2025,10]]}}}