{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T05:28:17Z","timestamp":1783402097742,"version":"3.54.6"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T00:00:00Z","timestamp":1732492800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T00:00:00Z","timestamp":1732492800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"name":"Provincial Postgraduate Innovation and Entrepreneurship Project of Anhui Province","award":["2022cxcysj020"],"award-info":[{"award-number":["2022cxcysj020"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62101206"],"award-info":[{"award-number":["62101206"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Process Lett"],"DOI":"10.1007\/s11063-024-11705-x","type":"journal-article","created":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T10:21:14Z","timestamp":1732530074000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Policy Optimization Algorithm with Activation Likelihood-Ratio for Multi-agent Reinforcement Learning"],"prefix":"10.1007","volume":"56","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7982-797X","authenticated-orcid":false,"given":"Lu","family":"Jia","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Binglin","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Du","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yewei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,25]]},"reference":[{"key":"11705_CR1","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals O, Babuschkin I, Czarnecki WM (2019) Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575:350\u2013354","journal-title":"Nature"},{"key":"11705_CR2","unstructured":"Berner C, Brockman G, Chan B, Cheung V, Debiak P, Dennison C, Farhi D, Fischer Q, Hashme S, Hesse C, Jozefowicz R, Gray S, Olsson C, Pachocki J, Petrov M, de Oliveira Pinto HP, Raiman J, Salimans T, Schlatter J, Schneider J, Sidor S, Sutskever I, Tang J, Wolski F, Zhang S (2019) Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680"},{"key":"11705_CR3","doi-asserted-by":"crossref","unstructured":"Ye D, Liu Z, Sun M, Shi B, Huang L (2020) Mastering complex control in MOBA games with deep reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 34(4), pp 6672\u20136679","DOI":"10.1609\/aaai.v34i04.6144"},{"issue":"1","key":"11705_CR4","doi-asserted-by":"publisher","first-page":"584","DOI":"10.1007\/s10458-023-09603-y","volume":"37","author":"A Smit","year":"2023","unstructured":"Smit A, Engelbrecht HA, Brink W, Pretorius A (2023) Scaling multi-agent reinforcement learning to full 11 versus 11 simulated robotic football. Auton Agents Multi-Agent Syst 37(1):584","journal-title":"Auton Agents Multi-Agent Syst"},{"key":"11705_CR5","doi-asserted-by":"crossref","unstructured":"Wang H, Tang H, Hao J, Hao X, Fu Y, Ma Y (2020) Large scale deep reinforcement learning in war-games. In: 2020 IEEE international conference on bioinformatics and biomedicine (BIBM), Seoul, Korea (South), pp 1693\u20131699","DOI":"10.1109\/BIBM49941.2020.9313387"},{"issue":"2","key":"11705_CR6","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1109\/TSMCC.2007.913919","volume":"38","author":"L Busoniu","year":"2008","unstructured":"Busoniu L, Babuska R, De Schutter B (2008) A comprehensive survey of multiagent reinforcement learning. IEEE Trans Syst Man Cybern C 38(2):156\u2013172","journal-title":"IEEE Trans Syst Man Cybern C"},{"key":"11705_CR7","unstructured":"Sunehag P, Lever G, Gruslys A, Czarnecki WM, Zambaldi VF, Jaderberg M, Lanctot M, Sonnerat N, Leibo JZ, Tuyls K, Graepel T (2018) Value-decomposition networks for cooperative multi-agent learning based on team reward. In: 17th international conference on autonomous agents and multiagent systems, vol 3, pp 2085\u20132087"},{"key":"11705_CR8","unstructured":"Rashid T, Samvelyan M, de Witt CS, Farquhar G, Foerster JN, Whiteson S (2018) QMIX: monotonic value function factorization for deep multi-agent reinforcement learning. In: Proceedings of the 35th international conference on machine learning, vol 80, pp 4295\u20134304"},{"key":"11705_CR9","unstructured":"Son K, Kim D, Kang WJ, Hostallero D, Yi Y (2019) QTRAN: learning to factorize with transformation for cooperative multi-agent reinforcement learning. In: International conference on machine learning, pp 5887\u20135896"},{"key":"11705_CR10","first-page":"6379","volume":"30","author":"R Lowe","year":"2017","unstructured":"Lowe R, Wu Y, Tamar A, Harb J, Abbeel P, Mordatch I (2017) Multiagent actor-critic for mixed cooperative-competitive environments. Adv Neural Inf Process Syst 30:6379\u20136390","journal-title":"Adv Neural Inf Process Syst"},{"key":"11705_CR11","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971"},{"key":"11705_CR12","doi-asserted-by":"crossref","unstructured":"Foerster JN, Farquhar G, Afouras T, Nardelli N, Whiteson S (2018) Counterfactual multiagent policy gradients. In: Proceedings of the AAAI conference on artificial intelligence, vol 32(1)","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"11705_CR13","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347"},{"key":"11705_CR14","unstructured":"Yu C, Velu A, Vinitsky E, Wang Y, Bayen AM, Wu Y (2022) The surprising effectiveness of MAPPO in cooperative, multi-agent games. arXiv preprint arXiv:2103.01955"},{"key":"11705_CR15","unstructured":"Wang Y, He H, Tan X (2020) Truly proximal policy optimization. arXiv preprint arXiv:1903.07940"},{"key":"11705_CR16","doi-asserted-by":"publisher","first-page":"96056","DOI":"10.1109\/ACCESS.2021.3094566","volume":"9","author":"W Zhu","year":"2021","unstructured":"Zhu W, Rosendo A (2021) A functional clipping approach for policy optimization algorithms. IEEE Access 9:96056\u201396063","journal-title":"IEEE Access"},{"key":"11705_CR17","unstructured":"Papoudakis G, Christianos F, Schafer L, Albrecht SV (2021) Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks. In: Thirty-fifth conference on neural information processing systems datasets and benchmarks track (round 1)"},{"key":"11705_CR18","unstructured":"de Witt CS, Gupta T, Makoviichuk D, Makoviychuk V, Torr PHS, Sun M, Whiteson S (2020) Is independent learning all you need in the starcraft multi-agent challenge? arXiv preprint arXiv:2011.09533"},{"key":"11705_CR19","unstructured":"Samvelyan M, Rashid T, de Witt CS, Farquhar G, Nardelli N, Rudner TGJ, Hung C-M, Torr PHS, Foerster JN, Whiteson S (2019) The starcraft multi-agent challenge. arXiv preprint arXiv:1902.04043"},{"key":"11705_CR20","unstructured":"Terry JK, Grammel N, Hari A, Santos L, Black B (2020) Revisiting parameter sharing in multi-agent deep reinforcement learning. arXiv preprint arXiv:2005.13625"},{"key":"11705_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10458-020-09455-w","volume":"34","author":"H Mao","year":"2020","unstructured":"Mao H, Zhang Z, Xiao Z, Gong Z, Ni Y (2020) Learning multi-agent communication with double attentional deep reinforcement learning. Auton Agent Multi-Agent Syst 34:1\u201334","journal-title":"Auton Agent Multi-Agent Syst"},{"key":"11705_CR22","doi-asserted-by":"publisher","first-page":"895","DOI":"10.1007\/s10462-021-09996-w","volume":"55","author":"S Gronauer","year":"2022","unstructured":"Gronauer S, Diepold K (2022) Multi-agent deep reinforcement learning: a survey. Artif Intell Rev 55:895\u2013943","journal-title":"Artif Intell Rev"},{"key":"11705_CR23","unstructured":"Schulman J, Levine S, Moritz P, Jordan MI, Abbeel P (2015) Trust region policy optimization. In: Proceedings of the 32nd international conference on machine learning, vol 37, pp 1889\u20131897"},{"issue":"4","key":"11705_CR24","doi-asserted-by":"publisher","first-page":"682","DOI":"10.1016\/j.neunet.2008.02.003","volume":"21","author":"J Peters","year":"2008","unstructured":"Peters J, Schaal S (2008) Reinforcement learning of motor skills with policy gradients. Neural Netw 21(4):682\u2013697","journal-title":"Neural Netw"},{"key":"11705_CR25","unstructured":"Marcin A, Anton R, Piotr S, Manu O, Sertan G, Raphael M, Leonard H, Matthieu G, Olivier P, Marcin M, Sylvain G, Olivier B (2021) What matters for on-policy deep actor-critic methods? A large-scale study. In: International conference on learning representations 2021"},{"key":"11705_CR26","unstructured":"Engstrom L, Ilyas A, Santurkar S, Tsipras D, Madry A (2020) Implementation matters in deep policy gradients: a case study on PPO and TRPO. arXiv preprint arXiv:2005.12729"},{"issue":"1\u20132","key":"11705_CR27","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/S0004-3702(98)00023-X","volume":"101","author":"LP Kaelbling","year":"1998","unstructured":"Kaelbling LP, Littman ML, Cassandra AR (1998) Planning and acting in partially observable stochastic domains. Artif Intell 101(1\u20132):99\u2013134","journal-title":"Artif Intell"},{"key":"11705_CR28","volume-title":"Exact and approximate algorithms for partially observable Markov decision processes","author":"AR Cassandra","year":"1998","unstructured":"Cassandra AR (1998) Exact and approximate algorithms for partially observable Markov decision processes. Brown University"},{"key":"11705_CR29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-27645-3","volume-title":"Reinforcement learning: state of the art","author":"M Wiering","year":"2012","unstructured":"Wiering M, Otterlo MV (2012) Reinforcement learning: state of the art. Springer. https:\/\/doi.org\/10.1007\/978-3-642-27645-3"},{"key":"11705_CR30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8","volume-title":"A concise introduction to decentralized POMDPs","author":"FA Oliehoek","year":"2016","unstructured":"Oliehoek FA, Amato C (2016) A concise introduction to decentralized POMDPs. Springer. https:\/\/doi.org\/10.1007\/978-3-319-28929-8"},{"key":"11705_CR31","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap TP, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. arXiv preprint arXiv:1602.01783"},{"key":"11705_CR32","first-page":"15032","volume":"34","author":"JK Terry","year":"2021","unstructured":"Terry JK, Black B, Hari A, Santos LS, Dieffendahl C, Williams NL, Lokesh Y, Horsch C, Ravi P (2021) Pettingzoo: gym for multi-agent reinforcement learning. Adv Neural Inf Process Syst 34:15032\u201315043","journal-title":"Adv Neural Inf Process Syst"},{"key":"11705_CR33","unstructured":"Terry JK, Black B, Hari A (2020) Supersuit: simple microwrappers for reinforcement learning environments. arXiv preprint arXiv:2008.08932"},{"key":"11705_CR34","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/s10458-021-09518-6","volume":"35","author":"L Pan","year":"2021","unstructured":"Pan L, Cai Q, Huang L (2021) Exploration in policy optimization through multiple paths. Auton Agent Multi-Agent Syst 35:33","journal-title":"Auton Agent Multi-Agent Syst"},{"key":"11705_CR35","unstructured":"Ilyas A, Engstrom L, Santurkar, S, Tsipras D, Janoos F, Rudolph L, Madry A (2018) Are deep policy gradient algorithms truly policy gradient algorithms? arXiv preprint arXiv:1811.02553"}],"container-title":["Neural Processing Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11063-024-11705-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11063-024-11705-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11063-024-11705-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,31]],"date-time":"2024-12-31T02:03:37Z","timestamp":1735610617000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11063-024-11705-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,25]]},"references-count":35,"journal-issue":{"issue":"6","published-online":{"date-parts":[[2024,12]]}},"alternative-id":["11705"],"URL":"https:\/\/doi.org\/10.1007\/s11063-024-11705-x","relation":{},"ISSN":["1573-773X"],"issn-type":[{"value":"1573-773X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,25]]},"assertion":[{"value":"5 November 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 November 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"247"}}