{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T22:53:25Z","timestamp":1784933605652,"version":"3.55.0"},"reference-count":38,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2021ZD0112904"],"award-info":[{"award-number":["2021ZD0112904"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1109\/tai.2024.3401649","type":"journal-article","created":{"date-parts":[[2024,5,15]],"date-time":"2024-05-15T13:47:08Z","timestamp":1715780828000},"page":"4984-4995","source":"Crossref","is-referenced-by-count":4,"title":["Bidirectional Influence and Interaction for Multiagent Reinforcement Learning"],"prefix":"10.1109","volume":"5","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9656-6918","authenticated-orcid":false,"given":"Shaoqi","family":"Sun","sequence":"first","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Processing, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Processing, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7587-8905","authenticated-orcid":false,"given":"Dawei","family":"Feng","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Processing, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1236-8318","authenticated-orcid":false,"given":"Bo","family":"Ding","sequence":"additional","affiliation":[{"name":"National Key Laboratory of Parallel and Distributed Processing, National University of Defense Technology, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2020.2977374"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ITSC.2014.6958095"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2906260"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s13218-020-00642-1"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-021-05961-4"},{"key":"ref6","article-title":"Exploration in deep reinforcement learning: A comprehensive survey","author":"Yang","year":"2021"},{"key":"ref7","article-title":"Cooperative-competitive reinforcement learning with history-dependent rewards","author":"He","year":"2020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-020-68447-8"},{"key":"ref9","first-page":"10707","article-title":"Shared experience actor-critic for multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Christianos","year":"2020"},{"key":"ref10","first-page":"2681","article-title":"Deep decentralized multi-task multi-agent reinforcement learning under partial observability","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Omidshafiei","year":"2017"},{"key":"ref11","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume":"99","author":"Ng","year":"1999","journal-title":"Proc. 16th Int. Conf. Machine Learn. (ICML)"},{"key":"ref12","article-title":"Investigating the impact of direct punishment on the emergence of cooperation in multi-agent reinforcement learning systems","author":"Dasgupta","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-024-09644-x"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/BF03037252"},{"key":"ref15","article-title":"A gentle introduction to the kernel distance","author":"Phillips","year":"2011"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1142\/S0129065704001899"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2014.2320500"},{"key":"ref18","first-page":"3991","article-title":"Celebrating diversity in shared multi-agent reinforcement learning","volume":"34","author":"Li","year":"2021","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1177\/105971230501300301"},{"issue":"1","key":"ref21","first-page":"7234","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref22","first-page":"6382","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Lowe","year":"2017"},{"key":"ref23","first-page":"24611","article-title":"The surprising effectiveness of PPO in cooperative multi-agent games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Yu","year":"2022"},{"key":"ref24","first-page":"23417","article-title":"Individual reward assisted multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2022"},{"key":"ref25","article-title":"Selectively sharing experiences improves multi-agent reinforcement learning","author":"Gerstgrasser","year":"2023"},{"key":"ref26","article-title":"Discovering causality for efficient cooperation in multi-agent environments","author":"Pina","year":"2023"},{"key":"ref27","first-page":"58","article-title":"EXPODE: Exploiting policy discrepancy for efficient exploration in multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Auton. Agents Multiagent Syst.","author":"Zhang","year":"2023"},{"key":"ref28","article-title":"Self-motivated multi-agent exploration","author":"Zhang","year":"2023"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2013.6760239"},{"key":"ref30","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref31","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"Schulman","year":"2015"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref33","article-title":"Mava: A research framework for distributed multi-agent reinforcement learning","author":"Pretorius","year":"2021"},{"key":"ref34","first-page":"8016","article-title":"Continuous coordination as a realistic scenario for lifelong learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Nekoei","year":"2021"},{"key":"ref35","first-page":"5824","article-title":"Gradient surgery for multi-task learning","volume":"33","author":"Yu","year":"2020","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/65"},{"key":"ref37","first-page":"5887","article-title":"QTRAN: Learning to factorize with transformation for cooperative multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Son","year":"2019"},{"key":"ref38","article-title":"Learning to share in multi-agent reinforcement learning","author":"Yi","year":"2022"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9078688\/10720652\/10531155.pdf?arnumber=10531155","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:09:22Z","timestamp":1755911362000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10531155\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10]]},"references-count":38,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/tai.2024.3401649","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10]]}}}