{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T18:13:51Z","timestamp":1764785631856,"version":"3.44.0"},"reference-count":81,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Science Foundation of China","doi-asserted-by":"publisher","award":["62276126","62250069"],"award-info":[{"award-number":["62276126","62250069"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1109\/tnnls.2025.3563773","type":"journal-article","created":{"date-parts":[[2025,5,6]],"date-time":"2025-05-06T13:03:56Z","timestamp":1746536636000},"page":"15807-15821","source":"Crossref","is-referenced-by-count":1,"title":["Learning to Coordinate With Different Teammates via Team Probing"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-4907-0304","authenticated-orcid":false,"given":"Hao","family":"Ding","sequence":"first","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengxing","family":"Jia","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9238-4747","authenticated-orcid":false,"given":"Zongzhang","family":"Zhang","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5198-9141","authenticated-orcid":false,"given":"Cong","family":"Guan","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8164-3410","authenticated-orcid":false,"given":"Feng","family":"Chen","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7803-0766","authenticated-orcid":false,"given":"Lei","family":"Yuan","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1732-9545","authenticated-orcid":false,"given":"Yang","family":"Yu","sequence":"additional","affiliation":[{"name":"National Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"A survey of progress on cooperative multi-agent reinforcement learning in open environment","author":"Yuan","year":"2023","journal-title":"arXiv:2312.01058"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1016\/j.energy.2018.04.042"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1007\/978-3-030-86514-6_29"},{"key":"ref4","first-page":"264","article-title":"SMARTS: Scalable multi-agent reinforcement learning training school for autonomous driving","volume-title":"Proc. Conf. Robot Learn.","author":"Zhou"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/tnnls.2022.3152251"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/TNNLS.2022.3158085"},{"key":"ref7","first-page":"20147","article-title":"Multi-agent dynamic algorithm configuration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xue"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1007\/s10489-022-04105-y"},{"key":"ref9","first-page":"4292","article-title":"Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rashid"},{"key":"ref10","first-page":"2085","article-title":"Value-decomposition networks for cooperative multi-agent learning based on team reward","volume-title":"Proc. 17th Int. Conf. Auto. Agents MultiAgent Syst.","author":"Sunehag"},{"key":"ref11","first-page":"6379","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. NIPS","author":"Lowe"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang","article-title":"DOP: Off-policy multi-agent decomposed policy gradients","key":"ref12"},{"key":"ref13","article-title":"Dealing with non-stationarity in multi-agent deep reinforcement learning","author":"Papoudakis","year":"2019","journal-title":"arXiv:1906.04737"},{"key":"ref14","first-page":"1989","article-title":"Scaling multi-agent reinforcement learning with selective parameter sharing","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Christianos"},{"key":"ref15","first-page":"24611","article-title":"The surprising effectiveness of PPO in cooperative, multi-agent games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kuba","article-title":"Trust region policy optimisation in multi-agent reinforcement learning","key":"ref16"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.1109\/CVPRW56347.2022.00022"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1007\/s11432-023-3853-y"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1007\/s11704-023-2733-5"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.1109\/TNNLS.2021.3129160"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1109\/TNNLS.2023.3236361"},{"key":"ref22","first-page":"16509","article-title":"Multi-agent reinforcement learning is a sequence modeling problem","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wen"},{"key":"ref23","first-page":"5175","article-title":"On the utility of learning about humans for human-AI coordination","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Carroll"},{"key":"ref24","first-page":"14502","article-title":"Collaborating with humans without human data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Strouse"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/ICRA40945.2020.9197197"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1016\/j.artint.2016.10.005"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1609\/aaai.v34i05.6196"},{"key":"ref28","first-page":"19210","article-title":"Agent modelling under partial observability for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Papoudakis"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.1609\/aaai.v24i1.7529"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhou","article-title":"Environment probing interaction policies","key":"ref30"},{"key":"ref31","first-page":"7920","article-title":"Fast adaptation to new environments via policy-dynamics value functions","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Raileanu"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1109\/TPAMI.2021.3079209"},{"key":"ref33","first-page":"1","article-title":"Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Papoudakis"},{"key":"ref34","first-page":"2186","article-title":"The StarCraft multi-agent challenge","volume-title":"Proc. Int. Conf. Auton. Agents Multiagent Syst. (AAMAS)","author":"Samvelyan"},{"doi-asserted-by":"publisher","key":"ref35","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref36","first-page":"517","article-title":"Exploiting locality of interaction in factored dec-POMDPs","volume-title":"Proc. Int. Conf. Auto. Agents Multiagent Syst.","author":"Oliehoek"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1109\/TNNLS.2022.3172572"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.1109\/TNNLS.2022.3191673"},{"doi-asserted-by":"publisher","key":"ref39","DOI":"10.1109\/TNNLS.2022.3216327"},{"doi-asserted-by":"publisher","key":"ref40","DOI":"10.1109\/TNNLS.2021.3121546"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.1109\/TNNLS.2023.3326744"},{"doi-asserted-by":"publisher","key":"ref42","DOI":"10.1109\/LRA.2022.3231497"},{"key":"ref43","first-page":"805","article-title":"Fictitious self-play in extensive-form games","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Heinrich"},{"key":"ref44","first-page":"4399","article-title":"\u2019Other-play\u2019 for zero-shot coordination for zero-shot coordination","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Hu"},{"key":"ref45","article-title":"Quasi-equivalence discovery for zero-shot emergent communication","author":"Bullard","year":"2021","journal-title":"arXiv:2103.08067"},{"key":"ref46","first-page":"4190","article-title":"A unified game-theoretic approach to multiagent reinforcement learning","volume-title":"Proc. Conf. Adv. Neural Inf. Process. Syst.","author":"Lanctot"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"M\u00fcller","article-title":"A generalized training approach for multiagent learning","key":"ref47"},{"doi-asserted-by":"publisher","key":"ref48","DOI":"10.1109\/TNNLS.2022.3156279"},{"doi-asserted-by":"publisher","key":"ref49","DOI":"10.1109\/TNNLS.2021.3105548"},{"doi-asserted-by":"publisher","key":"ref50","DOI":"10.1007\/978-3-031-20614-6_16"},{"doi-asserted-by":"publisher","key":"ref51","DOI":"10.1016\/j.jet.2005.12.010"},{"doi-asserted-by":"publisher","key":"ref52","DOI":"10.24963\/ijcai.2019\/78"},{"doi-asserted-by":"publisher","key":"ref53","DOI":"10.1145\/3627676.3627678"},{"doi-asserted-by":"publisher","key":"ref54","DOI":"10.1016\/j.artint.2016.02.004"},{"key":"ref55","first-page":"53","article-title":"Coordination and adaptation in impromptu teams","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Bowling"},{"doi-asserted-by":"publisher","key":"ref56","DOI":"10.1109\/robosoft51838.2021.9479430"},{"key":"ref57","first-page":"547","article-title":"Reasoning about hypothetical agent behaviours and their parameters","volume-title":"Proc. 16th Conf. Auto. Agents MultiAgent Syst.","author":"Albrecht"},{"doi-asserted-by":"publisher","key":"ref58","DOI":"10.1109\/TNNLS.2023.3262921"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Gu","article-title":"Online ad hoc teamwork under partial observability","key":"ref59"},{"doi-asserted-by":"publisher","key":"ref60","DOI":"10.1109\/TNNLS.2022.3222263"},{"doi-asserted-by":"publisher","key":"ref61","DOI":"10.1109\/TNNLS.2021.3128666"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Tang","article-title":"Discovering diverse multi-agent strategic behavior via reward randomization","key":"ref62"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhou","article-title":"Continuously discovering novel strategies via reward-switching policy optimization","key":"ref63"},{"key":"ref64","article-title":"Learning to cooperate with unseen agent via meta-reinforcement learning","author":"Charakorn","year":"2021","journal-title":"arXiv:2111.03431"},{"key":"ref65","first-page":"16899","article-title":"Dynamic population-based meta-learning for multi-agent communication with natural language","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gupta"},{"key":"ref66","first-page":"7204","article-title":"Trajectory diversity for zero-shot coordination","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lupu"},{"doi-asserted-by":"publisher","key":"ref67","DOI":"10.1609\/aaai.v37i5.25758"},{"doi-asserted-by":"publisher","key":"ref68","DOI":"10.1016\/j.artint.2018.01.002"},{"key":"ref69","first-page":"1804","article-title":"Opponent modeling in deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"He"},{"key":"ref70","first-page":"1802","article-title":"Learning policy representations in multiagent systems","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Grover"},{"key":"ref71","first-page":"1944","article-title":"Evaluating generalization in multiagent systems using agent-interaction graphs","volume-title":"Proc. Int. Conf. Auto. Agents Multiagent Syst.","author":"Grover"},{"key":"ref72","first-page":"1388","article-title":"A deep policy inference Q-network for multi-agent systems","volume-title":"Proc. Int. Conf. Auto. Agents Multiagent Syst.","author":"Hong"},{"doi-asserted-by":"publisher","key":"ref73","DOI":"10.1007\/978-3-319-28929-8"},{"key":"ref74","first-page":"243","article-title":"An alternative softmax operator for reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Asadi"},{"doi-asserted-by":"publisher","key":"ref75","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref76","article-title":"Layer normalization","author":"Ba","year":"2016","journal-title":"arXiv:1607.06450"},{"doi-asserted-by":"publisher","key":"ref77","DOI":"10.1016\/0377-0427(87)90125-7"},{"volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hausman","article-title":"Learning an embedding space for transferable robot skills","key":"ref78"},{"key":"ref79","first-page":"4767","article-title":"Multi-task reinforcement learning with soft modularization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref80","first-page":"5331","article-title":"Efficient off-policy meta-reinforcement learning via probabilistic context variables","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Rakelly"},{"doi-asserted-by":"publisher","key":"ref81","DOI":"10.1098\/rsta.2015.0202"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11151745\/10988880.pdf?arnumber=10988880","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,5]],"date-time":"2025-09-05T18:24:46Z","timestamp":1757096686000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10988880\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9]]},"references-count":81,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2025.3563773","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"type":"print","value":"2162-237X"},{"type":"electronic","value":"2162-2388"}],"subject":[],"published":{"date-parts":[[2025,9]]}}}