{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T23:00:22Z","timestamp":1784588422995,"version":"3.55.0"},"reference-count":216,"publisher":"Zhejiang University Press","issue":"4","license":[{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front Inform Technol Electron Eng"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1631\/fitee.2400259","type":"journal-article","created":{"date-parts":[[2025,5,7]],"date-time":"2025-05-07T09:45:51Z","timestamp":1746611151000},"page":"479-509","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Optimization methods in fully cooperative scenarios: a review of multiagent reinforcement learning","\u5b8c\u5168\u5408\u4f5c\u573a\u666f\u4e2d\u7684\u4f18\u5316\u65b9\u6cd5: \u591a\u667a\u80fd\u4f53\u5f3a\u5316\u5b66\u4e60\u7efc\u8ff0"],"prefix":"10.1631","volume":"26","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7873-2959","authenticated-orcid":false,"given":"Tao","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7240-7458","authenticated-orcid":false,"given":"Xinhao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinghan","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yulin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4913-5371","authenticated-orcid":false,"given":"Cheng","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongzhe","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"635","published-online":{"date-parts":[[2025,5,7]]},"reference":[{"issue":"6","key":"ref1","doi-asserted-by":"crossref","first-page":"3454","DOI":"10.1109\/TCOMM.2024.3365520","article-title":"Cooperative multi-agent learning for navigation via structured state abstraction","volume":"72","author":"Abdel-Aziz","year":"2024","journal-title":"IEEE Trans Commun"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1017\/s0263574799211174"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3319619.3321894"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ccwc57344.2023.10099211"},{"key":"ref5","article-title":"Emergent tool use from multi-agent autocurricula","volume-title":"Proc 8th Int Conf on Learning Representations","author":"Baker","year":"2020"},{"key":"ref6","article-title":"Unifying count-based exploration and intrinsic motivation","volume-title":"Proc 30th Conf on Neural Information Processing Systems","author":"Bellemare","year":"2016"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1287\/moor.27.4.819.297"},{"key":"ref8","article-title":"Weak-to-strong generalization: eliciting strong capabilities with weak supervision","volume-title":"Proc 41st Int Conf on Machine Learning","author":"Burns","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-023-48767-1"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/tits.2023.3330183"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.hcc.2023.100179"},{"key":"ref12","article-title":"On the utility of learning about humans for human-AI coordination","volume-title":"Proc 33rd Conf on Neural Information Processing Systems","author":"Carroll","year":"2019"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-63823-8_46"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2024.104697"},{"key":"ref15","first-page":"4996","article-title":"Redeeming intrinsic rewards via constrained optimization","volume-title":"Proc 36th Conf on Neural Information Processing Systems","author":"Chen","year":"2022"},{"key":"ref16","author":"Chen","year":"2023","journal-title":"Multi-agent consensus seeking via large language models"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i10.29011"},{"key":"ref18","article-title":"Contingency-aware exploration in reinforcement learning","volume-title":"Proc 7th Int Conf on Learning Representations","author":"Choi","year":"2019"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/access.2021.3110255"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1924-6"},{"key":"ref21","first-page":"1538","article-title":"TarMAC: targeted multi-agent communication","volume-title":"Proc 36th Int Conf on Machine Learning","author":"Das","year":"2019"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.65109\/JJTT8551"},{"key":"ref23","author":"de Witt","year":"2020","journal-title":"Is independent learning all you need in the StarCraft Multi-Agent Challenge?"},{"key":"ref24","author":"de Witt","year":"2021","journal-title":"Deep multiagent reinforcement learning for decentralized continuous cooperative control"},{"key":"ref25","first-page":"22069","article-title":"Learning individually inferred communication for multi-agent cooperation","volume-title":"Proc 34th Int Conf on Neural Information Processing Systems","author":"Ding","year":"2020"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3215774"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.123158"},{"key":"ref28","author":"ElSayed-Aly","year":"2022","journal-title":"Logic-based reward shaping for multi-agent reinforcement learning"},{"key":"ref29","article-title":"Diversity is all you need: learning skills without a reward function","volume-title":"Proc 7th Int Conf on Learning Representations","author":"Eysenbach","year":"2019"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.3390\/app12146938"},{"key":"ref31","author":"Foerster","year":"2016","journal-title":"Learning to communicate to solve riddles with deep distributed recurrent Q-networks"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref33","article-title":"DORA the Explorer: directed outreaching reinforcement action-selection","volume-title":"Proc 6th Int Conf on Learning Representations","author":"Fox","year":"2018"},{"key":"ref34","first-page":"6863","article-title":"Revisiting some common practices in cooperative multi-agent reinforcement learning","volume-title":"Proc 39th Int Conf on Machine Learning","author":"Fu","year":"2022"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/icsess47205.2019.9040781"},{"key":"ref36","volume-title":"A Primer in Game Theory","author":"Gibbons","year":"1992"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1142\/s2301385023410029"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/jiot.2022.3226953"},{"key":"ref39","first-page":"1311","article-title":"Automated curriculum learning for neural networks","volume-title":"Proc 34th Int Conf on Machine Learning","author":"Graves","year":"2017"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2023.103905"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/tii.2024.3391934"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10160923"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-71682-4_5"},{"key":"ref44","first-page":"1861","article-title":"Soft actorcritic: off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc 35th Int Conf on Machine Learning","author":"Haarnoja","year":"2018"},{"key":"ref45","article-title":"Finite-time convergence and sample complexity of multi-agent actor-critic reinforcement learning with average reward","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Hairi","year":"2022"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.65109\/OUOV7693"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2023.3236361"},{"key":"ref48","author":"Hao","year":"2022","journal-title":"Breaking the curse of dimensionality in multiagent state space: a unified agent permutation framework"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.14.3.159"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9628"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/iscc58397.2023.10217881"},{"key":"ref53","first-page":"12619","article-title":"Distributional reward estimation for effective multi-agent deep reinforcement learning","volume-title":"Proc 36th Conf on Neural Information Processing Systems","author":"Hu","year":"2022"},{"key":"ref54","author":"Hua","year":"2024","journal-title":"War and peace (WarA-gent): LLM-based multi-agent simulation of world wars"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-023-05197-w"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.12440"},{"key":"ref57","first-page":"3040","article-title":"Social influence as intrinsic motivation for multi-agent deep re-inforcement learning","volume-title":"Proc 36th Int Conf on Machine Learning","author":"Jaques","year":"2019"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2023.3335399"},{"key":"ref59","first-page":"10041","article-title":"MASER: multi-agent reinforcement learning with subgoals generated from experience replay buffer","volume-title":"Proc 39th Int Conf on Machine Learning","author":"Jeon","year":"2022"},{"key":"ref60","author":"Ji","year":"2024","journal-title":"AI alignment: a comprehensive survey"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-023-05058-6"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557292"},{"key":"ref63","first-page":"7265","article-title":"Learning attentional communication for multi-agent cooperation","volume-title":"Proc 32nd Int Conf on Neural Information Processing Systems","author":"Jiang","year":"2018"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i12.29196"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v18i1.21954"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.71330\/thenucleus.2023.1303"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.65109\/EHWR3444"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1049\/cth2.12413"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-6142-6_7"},{"key":"ref70","first-page":"13458","article-title":"Settling the variance of multi-agent policy gradients","volume-title":"Proc 35th Conf on Neural Information Processing Systems","author":"Kuba","year":"2021"},{"key":"ref71","article-title":"Trust region policy optimisation in multi-agent reinforcement learning","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Kuba","year":"2022"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5878"},{"key":"ref73","first-page":"4193","article-title":"A unified game-theoretic approach to multiagent reinforcement learning","volume-title":"Proc 31st Int Conf on Neural Information Processing Systems","author":"Lanctot","year":"2017"},{"key":"ref74","article-title":"In-context reinforcement learning with algorithm distillation","volume-title":"Proc 11th Int Conf on Learning Representations","author":"Laskin","year":"2023"},{"key":"ref75","article-title":"IMP-MARL: a suite of environments for large-scale infrastructure management planning via MARL","volume-title":"Proc 37th Int Conf on Neural Information Processing Systems","author":"Leroy","year":"2024"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.26028"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.65109\/TAMH6534"},{"key":"ref78","author":"Li","year":"2023","journal-title":"CAMEL: communicative agents for \u201cmind\u201d exploration of large language model society"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2023.3265358"},{"key":"ref80","first-page":"6346","article-title":"MURAL: meta-learning uncertainty-aware rewards for outcome-driven reinforcement learning","volume-title":"Proc 38th Int Conf on Machine Learning","author":"Li","year":"2021"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3190471"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.65109\/CZOY2835"},{"key":"ref83","first-page":"579","article-title":"AIIR-MIX: multiagent reinforcement learning meets attention individual intrinsic reward mixing network","volume-title":"Proc 14th Asian Conf on Machine Learning","author":"Li","year":"2023"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2023.3263013"},{"key":"ref85","article-title":"Dealing with non-stationarity in MARL via trust-region decomposition","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Li","year":"2022"},{"key":"ref86","first-page":"20470","article-title":"Cooperative open-ended learning framework for zero-shot coordination","volume-title":"Proc 40th Int Conf on Machine Learning","author":"Li","year":"2023a"},{"key":"ref87","author":"Li","year":"2023b","journal-title":"JiangJun: mastering Xiangqi by tackling non-transitivity in two-player zero-sum games"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.15884"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/tits.2024.3411487"},{"key":"ref90","first-page":"21937","article-title":"Lazy agents: a new perspective on solving sparse reward problem in multi-agent reinforcement learning","volume-title":"Proc 40th Int Conf on Machine Learning","author":"Liu","year":"2023"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-023-05551-2"},{"key":"ref92","first-page":"6826","article-title":"Cooperative exploration for multi-agent deep reinforcement learning","volume-title":"Proc 38th Int Conf on Machine Learning","author":"Liu","year":"2021"},{"key":"ref93","first-page":"206","article-title":"Exploration in model-based reinforcement learning by empirically estimating learning progress","volume-title":"Proc 25th Int Conf on Neural Information Processing Systems","author":"Lopes","year":"2012"},{"key":"ref94","first-page":"6382","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc 31st Int Conf on Neural Information Processing Systems","author":"Lowe","year":"2017"},{"key":"ref95","article-title":"EUREKA: human-level reward design via coding large language models","volume-title":"Proc 12th Int Conf on Learning Representations","author":"Ma","year":"2024"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5955"},{"key":"ref97","article-title":"MAVEN: multi-agent variational exploration","volume-title":"Proc 33rd Int Conf on Neural Information Processing Systems","author":"Mahajan","year":"2019"},{"key":"ref98","article-title":"Sample efficient deep re-inforcement learning via uncertainty estimation","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Mai","year":"2022"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-94662-3_3"},{"key":"ref100","author":"Mao","year":"2023","journal-title":"Transformer in Transformer as backbone for deep reinforcement learning"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.3390\/info14110623"},{"key":"ref102","article-title":"LIGS: learnable intrinsic-reward generation selection for multi-agent learning","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Mguni","year":"2022"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/tmlcn.2024.3368367"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref105","first-page":"376","article-title":"Dealing with non-stationarity in decentralized cooperative multi-agent deep reinforcement learning via multitimescale learning","volume-title":"Proc 2nd Conf on Lifelong Learning Agents","author":"Nekoei","year":"2023"},{"key":"ref106","first-page":"278","article-title":"Policy invariance under reward transformations: theory and application to reward shaping","volume-title":"Proc 16th Int Conf on Machine Learning","author":"Ng","year":"1999"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.65109\/XXYF8428"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-024-05293-5"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-022-04105-y"},{"key":"ref110","first-page":"2721","article-title":"Count-based exploration with neural density models","volume-title":"Proc 34th Int Conf on Machine Learning","author":"Ostrovski","year":"2017"},{"key":"ref111","first-page":"27862","article-title":"MATE: benchmarking multi-agent reinforcement learning in distributed target coverage control","volume-title":"Proc 36th Conf on Neural Information Processing Systems","author":"Pan","year":"2022"},{"key":"ref112","author":"Papoudakis","year":"2019","journal-title":"Dealing with non-stationarity in multi-agent deep reinforcement learning"},{"key":"ref113","article-title":"Bench-marking multi-agent deep reinforcement learning algorithms in cooperative tasks","volume-title":"Proc 35th Conf on Neural Information Processing Systems","author":"Papoudakis","year":"2022"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1145\/3586183.3606763"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/mcom.020.2300199"},{"key":"ref116","first-page":"5062","article-title":"Self-supervised exploration via disagreement","volume-title":"Proc 36th Int Conf on Machine Learning","author":"Pathak","year":"2019"},{"key":"ref117","first-page":"12208","article-title":"FACMAC: factored multi-agent centralised policy gradients","volume-title":"Proc 35th Conf on Neural Information Processing Systems","author":"Peng","year":"2021"},{"key":"ref118","author":"Peng","year":"2017","journal-title":"Multiagent bidirectionally-coordinated nets: emergence of human-level coordination in learning to play StarCraft combat games"},{"key":"ref119","author":"Perez-Liebana","year":"2019","journal-title":"The multi-agent reinforcement learning in Malm\u00d6 (MARL\u00d6) competition"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1126\/science.add4679"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-019-05864-5"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3146858"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121252"},{"key":"ref124","article-title":"Scalable multiagent reinforcement learning for networked systems with average reward","volume-title":"Proc 34th Int Conf on Neural Information Processing Systems","author":"Qu","year":"2020"},{"key":"ref125","article-title":"Hokoff: real game dataset from Honor of Kings and its offline reinforcement learning benchmarks","volume-title":"Proc 37th Int Conf on Neural Information Processing Systems","author":"Qu","year":"2024"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-019-09433-x"},{"key":"ref127","first-page":"4295","article-title":"QMIX: monotonic value function factorisation for deep multiagent reinforcement learning","volume-title":"Proc 35th Int Conf on Machine Learning","author":"Rashid","year":"2018"},{"key":"ref128","article-title":"Weighted QMIX: expanding monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"Proc 34th Int Conf on Neural Information Processing Systems","author":"Rashid","year":"2020"},{"key":"ref129","article-title":"Implicit generative modeling for efficient exploration","volume-title":"Proc 37th Int Conf on Machine Learning","author":"Ratzlaff","year":"2020"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121111"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/tii.2024.3413356"},{"key":"ref132","article-title":"Pommerman: a multi-agent playground","volume-title":"Proc 14th AAAI Conf on Artificial Intelligence and Interactive Digital Entertainment","author":"Resnick","year":"2018"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2023.104308"},{"key":"ref134","author":"Roostaie","year":"2021","journal-title":"EnTRPO: trust region policy optimization method with entropy regularization"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.65109\/lvzz5205"},{"key":"ref136","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc 32nd Int Conf on Machine Learning","author":"Schulman","year":"2015"},{"key":"ref137","article-title":"Self-organized group for cooperative multi-agent reinforcement learning","volume-title":"Proc 36th Int Conf on Neural Information Processing Systems","author":"Shao","year":"2022"},{"key":"ref138","article-title":"Dynamics-aware unsupervised discovery of skills","volume-title":"Proc 8th Int Conf on Learning Representations","author":"Sharma","year":"2020"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.65109\/YBXD5829"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/466"},{"key":"ref141","article-title":"ResQ: a residual Q function-based approach for multi-agent reinforcement learning value factorization","volume-title":"Proc 36th Int Conf on Neural Information Processing Systems","author":"Shen","year":"2022"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2020.102738"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.1023\/a:1007678930559"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.21236\/ADA440280"},{"key":"ref145","first-page":"5887","article-title":"QTRAN: learning to factorize with transformation for cooperative multi-agent reinforcement learning","volume-title":"Proc 36th Int Conf on Machine Learning","author":"Son","year":"2019"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.65109\/FMRF8293"},{"key":"ref147","first-page":"2252","article-title":"Learning multiagent communication with backpropagation","volume-title":"Proc 30th Int Conf on Neural Information Processing Systems","author":"Sukhbaatar","year":"2016"},{"key":"ref148","author":"Sunehag","year":"2017","journal-title":"Value-decomposition networks for cooperative multi-agent learning"},{"key":"ref149","article-title":"Temporal Credit Assignment in Reinforcement Learning","author":"Sutton","year":"1984","journal-title":"University of Massachusetts Amherst, Massachusetts, USA"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1007\/BF00115009"},{"key":"ref151","first-page":"2750","article-title":"#Exploration: a study of count-based exploration for deep reinforcement learning","volume-title":"Proc 31st Int Conf on Neural Information Processing Systems","author":"Tang","year":"2017"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i10.26379"},{"key":"ref153","author":"Vanneste","year":"2020","journal-title":"Learning to communicate using counterfactual reasoning"},{"key":"ref154","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112000"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.1016\/j.cja.2024.03.030"},{"key":"ref156","article-title":"Multi-agent rein-forcement learning for active voltage control on power distribution networks","volume-title":"Proc 35th Int Conf on Neural Information Processing Systems","author":"Wang","year":"2021a"},{"key":"ref157","article-title":"QPLEX: duplex dueling multi-agent Q-learning","volume-title":"Proc 9th Int Conf on Learning Representations","author":"Wang","year":"2021b"},{"key":"ref158","first-page":"29142","article-title":"Towards understanding cooperative multi-agent Q-learning with value factorization","volume-title":"Proc 35th Conf on Neural Information Processing Systems","author":"Wang","year":"2021c"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.1109\/jas.2022.105506"},{"key":"ref160","first-page":"23417","article-title":"Individual reward assisted multi-agent reinforcement learning","volume-title":"Proc 39th Int Conf on Machine Learning","author":"Wang","year":"2022"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.3390\/math10152728"},{"key":"ref162","article-title":"Learning nearly decomposable value functions via communication minimization","volume-title":"Proc 8th Int Conf on Learning Representations","author":"Wang","year":"2020a"},{"key":"ref163","first-page":"9876","article-title":"ROMA: multi-agent reinforcement learning with emergent roles","volume-title":"Proc 37th Int Conf on Machine Learning","author":"Wang","year":"2020b"},{"key":"ref164","article-title":"RODE: learning roles to decompose multi-agent tasks","volume-title":"Proc 9th Int Conf on Learning Representations","author":"Wang","year":"2021"},{"key":"ref165","article-title":"Context-aware sparse deep coordination graphs","volume-title":"Proc 10th Int Conf on Learning Representations","author":"Wang","year":"2022"},{"key":"ref166","article-title":"Action semantics network: considering the effects of actions in multiagent systems","volume-title":"Proc 8th Int Conf on Learning Representations","author":"Wang","year":"2020"},{"key":"ref167","article-title":"Order matters: agent-by-agent policy optimization","volume-title":"Proc 11th Int Conf on Learning Representations","author":"Wang","year":"2023"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.65109\/WOCO9539"},{"key":"ref169","first-page":"1085","article-title":"Evaluating the perceived safety of urban city via maximum entropy deep inverse reinforcement learning","volume-title":"Proc 14th Asian Conf on Machine Learning","author":"Wang","year":"2023"},{"key":"ref170","author":"Wang","year":"2024","journal-title":"Describe, explain, plan and select: interactive planning with large language models enables open-world multi-task agents"},{"key":"ref171","article-title":"Multi-agent re-inforcement learning is a sequence modeling problem","volume-title":"Proc 36th Int Conf on Neural Information Processing Systems","author":"Wen","year":"2022"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.1016\/0377-2217(89)90348-2"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1190"},{"key":"ref174","first-page":"792","article-title":"Principled methods for advising reinforcement learning agents","volume-title":"Proc 20th Int Conf on Machine Learning","author":"Wiewiora","year":"2003"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1109\/tvt.2020.2997896"},{"key":"ref176","first-page":"26437","article-title":"Coordinated proximal policy optimization","volume-title":"Proc 35th Conf on Neural Information Processing Systems","author":"Wu","year":"2021"},{"key":"ref177","doi-asserted-by":"publisher","DOI":"10.65109\/TVAW9016"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1109\/tvt.2023.3312574"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.11.059"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i10.26384"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1109\/tsg.2020.2971427"},{"key":"ref182","author":"Xu","year":"2024","journal-title":"Exploring large language models for communication games: an empirical study on Werewolf"},{"key":"ref183","first-page":"73898","article-title":"Dual self-awareness value decomposition framework without individual global max for cooperative MARL","volume-title":"Proc 37th Conf on Neural Information Processing Systems","author":"Xu","year":"2023"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1109\/tte.2023.3236324"},{"key":"ref185","article-title":"An efficient transfer learning framework for multiagent reinforcement learning","volume-title":"Proc 35th Int Conf on Neural Information Processing Systems","author":"Yang","year":"2021"},{"key":"ref186","article-title":"Multi-agent determinantal Q-learning","volume-title":"Proc 37th Int Conf on Machine Learning","author":"Yang","year":"2020a"},{"key":"ref187","author":"Yang","year":"2020b","journal-title":"Qatten: a general framework for cooperative multiagent reinforcement learning"},{"key":"ref188","article-title":"Q-value path decomposition for deep multiagent reinforcement learning","volume-title":"Proc 37th Int Conf on Machine Learning","author":"Yang","year":"2020c"},{"key":"ref189","author":"Yang","year":"2022","journal-title":"When to go, and when to explore: the benefit of post-exploration in intrinsic motivation"},{"key":"ref190","author":"Ye","year":"2023","journal-title":"Towards global optimality in cooperative MARL with the transformation and distillation framework"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2019.00474"},{"key":"ref192","article-title":"Learning to share in multi-agent reinforcement learning","volume-title":"Proc 36th Int Conf on Neural Information Processing Systems","author":"Yi","year":"2022"},{"key":"ref193","article-title":"The surprising effectiveness of PPO in cooperative multi-agent games","volume-title":"Proc 36th Int Conf on Neural Information Processing Systems","author":"Yu","year":"2022"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i10.26389"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2300548"},{"key":"ref196","first-page":"46105","article-title":"Automatic grouping for efficient cooperative multi-agent reinforcement learning","volume-title":"Proc 37th Conf on Neural Information Processing Systems","author":"Zang","year":"2023"},{"key":"ref197","first-page":"278","article-title":"Learning to coordinate in multi-agent systems: a coordinated actor-critic algorithm and finite-time guarantees","volume-title":"Proc 4th Annual Learning for Dynamics and Control Conf","author":"Zeng","year":"2022"},{"key":"ref198","first-page":"12333","article-title":"DouZero: mastering DouDizhu with self-play deep reinforcement learning","volume-title":"Proc 38th Int Conf on Machine Learning","author":"Zha","year":"2021"},{"key":"ref199","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599379"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.1900661"},{"key":"ref201","article-title":"GoBigger: a scalable platform for cooperative-competitive multi-agent interactive simulation","volume-title":"Proc 11th Int Conf on Learning Representations","author":"Zhang","year":"2023"},{"key":"ref202","doi-asserted-by":"publisher","DOI":"10.1016\/j.jmsy.2023.08.011"},{"key":"ref203","author":"Zhang","year":"2020","journal-title":"Multi-agent collaboration via reward attribution decomposition"},{"key":"ref204","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i13.17357"},{"key":"ref205","first-page":"2465","article-title":"Fast teammate adaptation in the presence of sudden policy change","volume-title":"Proc 39th Conf on Uncertainty in Artificial Intelligence","author":"Zhang","year":"2023"},{"key":"ref206","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2100594"},{"issue":"12","key":"ref207","first-page":"14","article-title":"Survey of fully cooperative multi-agent deep reinforcement learning","volume":"59","author":"Zhao","year":"2023","journal-title":"Comput Eng Appl"},{"key":"ref208","doi-asserted-by":"publisher","DOI":"10.1109\/cog51982.2022.9893576"},{"key":"ref209","article-title":"Episodic multi-agent reinforcement learning with curiosity-driven exploration","volume-title":"Proc 35th Int Conf on Neural Information Processing Systems","author":"Zheng","year":"2021"},{"key":"ref210","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11371"},{"key":"ref211","doi-asserted-by":"publisher","DOI":"10.1109\/ase.2019.00077"},{"key":"ref212","doi-asserted-by":"publisher","DOI":"10.1016\/j.cja.2024.04.008"},{"key":"ref213","author":"Zhu","year":"2023","journal-title":"Ghost in the Minecraft: generally capable agents for open-world environments via large language models with text-based knowledge and memory"},{"key":"ref214","article-title":"Behavior proximal policy optimization","volume-title":"Proc 11th Int Conf on Learning Representations","author":"Zhuang","year":"2023"},{"key":"ref215","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20915"},{"key":"ref216","author":"Zou","year":"2019","journal-title":"Reward shaping via meta-learning"}],"container-title":["Frontiers of Information Technology &amp; Electronic Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1631\/FITEE.2400259.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1631\/FITEE.2400259\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1631\/FITEE.2400259.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T06:37:21Z","timestamp":1771655841000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1631\/FITEE.2400259"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4]]},"references-count":216,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2049"],"URL":"https:\/\/doi.org\/10.1631\/fitee.2400259","relation":{},"ISSN":["2095-9184","2095-9230"],"issn-type":[{"value":"2095-9184","type":"print"},{"value":"2095-9230","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4]]},"assertion":[{"value":"6 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"All the authors declare that they have no conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}