{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T05:00:12Z","timestamp":1768280412678,"version":"3.49.0"},"reference-count":106,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62406270"],"award-info":[{"award-number":["62406270"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"STCSM Shanghai Rising-Star Program","award":["24YF2748800"],"award-info":[{"award-number":["24YF2748800"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62231019"],"award-info":[{"award-number":["62231019"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010877","name":"Science, Technology and Innovation Commission of Shenzhen Municipality","doi-asserted-by":"publisher","award":["2024SC0010"],"award-info":[{"award-number":["2024SC0010"]}],"id":[{"id":"10.13039\/501100010877","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72495131"],"award-info":[{"award-number":["72495131"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1109\/tpami.2025.3620954","type":"journal-article","created":{"date-parts":[[2025,10,13]],"date-time":"2025-10-13T17:40:39Z","timestamp":1760377239000},"page":"1657-1673","source":"Crossref","is-referenced-by-count":0,"title":["Learning Roles With Emergent Social Value Orientations"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2985-1098","authenticated-orcid":false,"given":"Wenhao","family":"Li","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3064-5128","authenticated-orcid":false,"given":"Xiangfeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Key Laboratory of Mathematics and Engineering Applications (MoE) and Shanghai Institute of AI for Education, East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1028-5989","authenticated-orcid":false,"given":"Bo","family":"Jin","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7640-9759","authenticated-orcid":false,"given":"Jingyi","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Psychology and Cognitive Science, East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7493-0911","authenticated-orcid":false,"given":"Hongyuan","family":"Zha","sequence":"additional","affiliation":[{"name":"School of Data Science, The Chinese University of Hong Kong, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature02043"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1126\/science.325_1196"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1098\/rstb.2020.0291"},{"key":"ref4","article-title":"Open problems in cooperative AU","author":"Dafoe","year":"2020"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.3998\/mpub.20269"},{"key":"ref6","first-page":"269","article-title":"Cooperation with bottom-up reputation dynamics","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Xu","year":"2019"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1146\/annurev.soc.24.1.183"},{"key":"ref8","first-page":"6382","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Lowe","year":"2017"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1126\/science.aau6249"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00941"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3102140"},{"key":"ref13","first-page":"25584","article-title":"Information design in multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Lin","year":"2023"},{"key":"ref14","article-title":"Many-agent reinforcement learning","author":"Yang","year":"2021"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103535"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-022-09575-5"},{"key":"ref17","volume-title":"Theory of Games and Economic Behavior (Commemorative Edition)","author":"von Neumann","year":"2007"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-017-5040-0"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1654.001.0001"},{"key":"ref20","first-page":"464","article-title":"Multi-agent reinforcement learning in sequential social dilemmas","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Leibo","year":"2017"},{"key":"ref21","first-page":"3330","article-title":"Inequity aversion improves cooperation in intertemporal social dilemmas","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hughes","year":"2018"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511807763"},{"key":"ref23","article-title":"Reward redistribution mechanisms in multi-agent reinforcement learning","volume-title":"Proc. AAMAS Adaptive Learn. Agents Workshop","author":"Ibrahim","year":"2020"},{"key":"ref24","first-page":"1887","article-title":"Silly rules improve the capacity of agents to learn stable enforcement and compliance behaviors","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Koster","year":"2020"},{"key":"ref25","first-page":"789","article-title":"Gifting in multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Lupu","year":"2020"},{"key":"ref26","first-page":"15208","article-title":"Learning to incentivize other learning agents","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Yang","year":"2020"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1177\/26339137231162025"},{"key":"ref28","article-title":"Birds of a feather flock together: A close look at cooperation emergence via multi-agent Rl","author":"Dong","year":"2021"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1126\/science.7466396"},{"key":"ref30","article-title":"Learning reciprocity in complex sequential social dilemmas","author":"Eccles","year":"2019"},{"key":"ref31","first-page":"115","article-title":"Cooperation and reputation dynamics with reinforcement learning","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Anastassacos","year":"2021"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1037\/0022-3514.54.5.811"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1086\/293213"},{"key":"ref34","first-page":"9983","article-title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Pretorius","year":"2020"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.4324\/9780203808498.ch12"},{"key":"ref36","first-page":"217","article-title":"Other-regarding preferences","volume-title":"The Handbook of Experimental Economics","author":"Cooper","year":"2016"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1006\/game.1996.0081"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/0191-8869(81)90084-2"},{"issue":"2","key":"ref39","first-page":"156","article-title":"Altruism and economics","volume":"83","author":"Simon","year":"1993","journal-title":"Amer. Econ. Rev."},{"key":"ref40","first-page":"869","article-title":"Social diversity and social preferences in mixed-motive reinforcement learning","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"McKee","year":"2020"},{"key":"ref41","article-title":"Prisoners\u2019 dilemma","year":"2022"},{"key":"ref42","volume-title":"Interpersonal relations: A theory of interdependence","author":"Kelley","year":"1978"},{"key":"ref43","first-page":"2043","article-title":"Prosocial learning agents solve generalized stag hunts better than selfish ones","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Peysakhovich","year":"2018"},{"key":"ref44","first-page":"3330","article-title":"Inequity aversion improves cooperation in intertemporal social dilemmas","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hughes","year":"2018"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-019-09411-3"},{"key":"ref46","first-page":"683","article-title":"Evolving intrinsic motivations for altruistic behavior","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Wang","year":"2019"},{"key":"ref47","first-page":"15786","article-title":"Emergent reciprocity and team formation from randomized uncertain social preferences","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Baker","year":"2020"},{"key":"ref48","first-page":"498","article-title":"D3C: Reducing the price of anarchy in multi-agent learning","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Gemp","year":"2022"},{"key":"ref49","first-page":"15119","article-title":"Learning to share in multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Yi","year":"2022"},{"key":"ref50","first-page":"1536","article-title":"Balancing rational and other-regarding preferences in cooperative-competitive environments","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Ivanov","year":"2021"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/54"},{"key":"ref52","article-title":"Model-free conventions in multi-agent reinforcement learning with heterogeneous preferences","author":"K\u00f6ster","year":"2020"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1038\/380121a0"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/j.anbehav.2005.03.004"},{"key":"ref55","volume-title":"The Condensed Wealth of Nations","author":"Butler","year":"2012"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-6494.2009.00615.x"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1111\/joop.12332"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1561\/9781680837896"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511571121"},{"key":"ref60","article-title":"Consequentialist conditional cooperation in social dilemmas with imperfect information","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Peysakhovich","year":"2018"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6190"},{"key":"ref62","first-page":"709","article-title":"Dynamic programming for partially observable stochastic games","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Hansen","year":"2004"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2006.884615"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-05990-7_20"},{"issue":"4","key":"ref65","first-page":"1","article-title":"A road map to the future for the auto industry","author":"Gao","year":"2014","journal-title":"McKinsey Quart."},{"key":"ref66","article-title":"Cooperation promotion multi-agent reinforcement learning","author":"Li","year":"2022"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1126\/science.aaf2654"},{"issue":"2","key":"ref68","first-page":"39","article-title":"An overview of the policy and legal aspects of the international climate change regime (Part I)","volume":"9","author":"Grimeaud","year":"2001","journal-title":"Environ. Liability"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.3386\/w6199"},{"key":"ref70","first-page":"265","article-title":"TensorFlow: A system for large-scale machine learning","volume-title":"Proc. USENIX Conf. Operating Syst. Des. Implementation","author":"Abadi","year":"2016"},{"key":"ref71","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Eysenbach","year":"2019"},{"key":"ref72","first-page":"9876","article-title":"ROMA: Multi-agent reinforcement learning with emergent roles","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2020"},{"key":"ref73","first-page":"15084","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Chen","year":"2021"},{"key":"ref74","first-page":"11213","article-title":"Oracles & followers: Stackelberg equilibria in deep multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Gerstgrasser","year":"2023"},{"key":"ref75","first-page":"122","article-title":"Learning with opponent-learning awareness","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Foerster","year":"2018"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1208417109"},{"key":"ref77","volume-title":"The Selfish Gene","author":"Davis","year":"2017"},{"key":"ref78","article-title":"Lance Armstrong and the prisoners\u2019 dilemma of doping in professional sports","author":"Schneier","year":"2012","journal-title":"Wired"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1016\/0165-4896(84)90022-2"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1126\/sciadv.abk2607"},{"key":"ref81","first-page":"1436","article-title":"Adaptive incentive design with multi-agent meta-gradient reinforcement learning","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Yang","year":"2022"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/614"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-022-09580-8"},{"key":"ref84","article-title":"Correcting experience replay for multi-agent communication","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ahilan","year":"2021"},{"key":"ref85","first-page":"77","article-title":"A generalised method for empirical game theoretic analysis","volume-title":"Proc. Int. Conf. Auton. Agents MultiAgent Syst.","author":"Tuyls","year":"2018"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1017\/cbo9781107297067.027"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1023\/A:1010071910869"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-44564-1_12"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1145\/544741.544749"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45023-8_38"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1007\/11559221_19"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.4018\/978-1-59904-510-8.ch012"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-22636-6_7"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1504\/IJAOSE.2010.036984"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-39975-6_3"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2996209"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1142\/S0218194018500043"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1023\/B:AGNT.0000018806.20944.ef"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v24i1.7679"},{"key":"ref100","article-title":"RODE: Learning roles to decompose multi-agent tasks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2021"},{"key":"ref101","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Sutton","year":"1999"},{"key":"ref102","article-title":"An open source implementation of sequential social dilemma games","author":"Vinitsky","year":"2019"},{"key":"ref103","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2015"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref105","volume-title":"Differential Equations, Dynamical Systems, and an Introduction to Chaos","author":"Hirsch","year":"2012"},{"key":"ref106","article-title":"On differentiating parameterized argmin and argmax problems with application to bi-level optimization","author":"Gould","year":"2016"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11345188\/11202682.pdf?arnumber=11202682","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,12]],"date-time":"2026-01-12T22:01:01Z","timestamp":1768255261000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11202682\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":106,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3620954","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2]]}}}