{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T16:23:15Z","timestamp":1782404595905,"version":"3.54.5"},"reference-count":69,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/access.2024.3497322","type":"journal-article","created":{"date-parts":[[2024,11,13]],"date-time":"2024-11-13T18:57:40Z","timestamp":1731524260000},"page":"167452-167470","source":"Crossref","is-referenced-by-count":7,"title":["Multiagent Deep Reinforcement Learning Algorithms in StarCraft II: A Review"],"prefix":"10.1109","volume":"12","author":[{"given":"Yanyan","family":"Li","sequence":"first","affiliation":[{"name":"School of Automation, Central South University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yijun","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Automation, Central South University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6020-3086","authenticated-orcid":false,"given":"Yiwei","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Automation, Central South University, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/tcyb.2020.2977374"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1147\/rd.441.0206"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2017.2737145"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2019.2896986"},{"issue":"12","key":"ref5","first-page":"2537","article-title":"Deep multi-agent reinforcement learning: A survey","volume":"46","author":"Xing-Xing","year":"2020","journal-title":"Acta Automatica Sinica"},{"key":"ref6","article-title":"MSC: A dataset for macro-management in StarCraft II","author":"Wu","year":"2017","journal-title":"arXiv:1710.03131"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2023.3265975"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.3778\/j.issn.1002-8331.1912-0100"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2023.3329763"},{"issue":"1","key":"ref10","first-page":"351","article-title":"Playing Atari with deep reinforcement learning","volume":"21","author":"Mnih","year":"2013","journal-title":"Comput. Sci."},{"key":"ref11","article-title":"Deep recurrent Q-learning for partially observable MDPs","author":"Hausknecht","year":"2015","journal-title":"arXiv:1507.06527"},{"key":"ref12","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref13","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv:1509.02971"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"issue":"8","key":"ref15","first-page":"1","article-title":"Overview on multi-agent reinforcement learning","volume":"46","author":"Du","year":"2019","journal-title":"Comput. Sci."},{"issue":"4","key":"ref16","first-page":"646","article-title":"A survey on multi-agent hierarchical reinforcement learning","volume":"15","author":"Yin","year":"2020","journal-title":"CAAI Trans. Intell. Syst."},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/tg.2022.3196925"},{"key":"ref18","article-title":"Revisiting some common practices in cooperative multi-agent reinforcement learning","author":"Fu","year":"2022","journal-title":"arXiv:2206.07505"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1117\/12.2585808"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-023-09633-6"},{"issue":"21","key":"ref21","first-page":"1","article-title":"Overview on reinforcement learning of multi-agent game","volume":"57","author":"Wang","year":"2021","journal-title":"Comput. Eng. Appl."},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SSCI.2018.8628682"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2021.3049539"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CAC53003.2021.9727882"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3266652"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CEC55065.2022.9870230"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2015.2509646"},{"key":"ref28","article-title":"Safe, multi-agent, reinforcement learning for autonomous driving","author":"Shalev-Shwartz","year":"2016","journal-title":"arXiv:1610.03295"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219993"},{"key":"ref30","article-title":"Guided deep reinforcement learning for swarm systems","author":"H\u00fcttenrauch","year":"2017","journal-title":"arXiv:1709.06011"},{"issue":"7","key":"ref31","first-page":"1610","article-title":"Research on multi-aircraft cooperative air combat method based on deep reinforcement learning","volume":"47","author":"Wei","year":"2021","journal-title":"Acta Automatica Sinica"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.5220\/0006393400170026"},{"issue":"5","key":"ref33","first-page":"115","article-title":"Target distribution model in cooperative air combat under uncertain environment","volume":"45","author":"Ou","year":"2020","journal-title":"Fire Control Command Control"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2017.2650872"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/PESGM40551.2019.8974083"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref37","article-title":"StarCraft II: A new challenge for reinforcement learning","author":"Vinyals","year":"2017","journal-title":"arXiv:1708.04782"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2019.8847963"},{"key":"ref39","article-title":"Reinforcement learning for real-time strategy games","author":"Germain","year":"2019"},{"key":"ref40","article-title":"An overview of multi-agent reinforcement learning from game theoretical perspective","author":"Yang","year":"2020","journal-title":"arXiv:2011.00583"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICAIIC54071.2022.9722618"},{"issue":"1","key":"ref42","first-page":"67","article-title":"A review of deep reinforcement learning theory and application","volume":"32","author":"Wan","year":"2019","journal-title":"Pattern Recognit. Artif. Intell."},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/MECO58584.2023.10155066"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref45","article-title":"Learning from delayed rewards","author":"Watkins","year":"1989"},{"key":"ref46","volume-title":"On-line Q-learning using connectionist systems","author":"Rummery","year":"1994"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2023.3347128"},{"issue":"7","key":"ref48","first-page":"1301","article-title":"Important scientific problems of multi-agent deep reinforcement learning","volume":"46","author":"Sun","year":"2020","journal-title":"Acta Automatica Sinica"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2022.105506"},{"key":"ref50","article-title":"TorchCraft: A library for machine learning research on real-time strategy games","author":"Synnaeve","year":"2016","journal-title":"arXiv:1611.00625"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA52953.2021.00199"},{"key":"ref52","article-title":"The StarCraft multi-agent challenge","author":"Samvelyan","year":"2019","journal-title":"arXiv:1902.04043"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0172395"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"ref55","article-title":"Value-decomposition networks for cooperative multi-agent learning","author":"Sunehag","year":"2017","journal-title":"arXiv:1706.05296"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/icist.2018.8426160"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v15i1.5230"},{"key":"ref58","article-title":"Distributed prioritized experience replay","author":"Horgan","year":"2018","journal-title":"arXiv:1803.00933"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.3390\/s21103332"},{"key":"ref61","first-page":"1","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref63","article-title":"QTRAN: Learning to factorize with transformation for cooperative multi-agent reinforcement learning","author":"Son","year":"2019","journal-title":"arXiv:1905.05408"},{"key":"ref64","article-title":"Learning multiagent communication with backpropagation","author":"Sukhbaatar","year":"2016","journal-title":"arXiv:1605.07736"},{"key":"ref65","article-title":"Multiagent bidirectionally-coordinated nets: Emergence of human-level coordination in learning to play StarCraft combat games","author":"Peng","year":"2017","journal-title":"arXiv:1703.10069"},{"key":"ref66","article-title":"Revisiting the master-slave architecture in multi-agent deep reinforcement learning","author":"Kong","year":"2017","journal-title":"arXiv:1712.07305"},{"key":"ref67","article-title":"Qatten: A general framework for cooperative multiagent reinforcement learning","author":"Yang","year":"2020","journal-title":"arXiv:2002.03939"},{"key":"ref68","article-title":"QPLEX: Duplex dueling multi-agent Q-learning","author":"Wang","year":"2020","journal-title":"arXiv:2008.01062"},{"key":"ref69","first-page":"11853","article-title":"Learning implicit credit assignment for cooperative multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Information Process. Syst.","author":"Zhou"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10380310\/10752399.pdf?arnumber=10752399","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T16:14:07Z","timestamp":1732724047000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10752399\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":69,"URL":"https:\/\/doi.org\/10.1109\/access.2024.3497322","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}