{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T16:35:14Z","timestamp":1774974914293,"version":"3.50.1"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010815","name":"Anhui Provincial Development and Reform Commission","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010815","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1109\/smc52423.2021.9659219","type":"proceedings-article","created":{"date-parts":[[2022,1,6]],"date-time":"2022-01-06T20:34:35Z","timestamp":1641501275000},"page":"1178-1185","source":"Crossref","is-referenced-by-count":4,"title":["Distributed Reinforcement Learning with Self-Play in Parameterized Action Space"],"prefix":"10.1109","author":[{"given":"Jun","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shunyi","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangda","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiakai","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianmin","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref11","article-title":"Intrinsic motivation and automatic curricula via asymmetric self-play","author":"sukhbaatar","year":"2018","journal-title":"Proceedings of the 6th International Conference on Learning Representations (ICLR-2018)"},{"key":"ref12","article-title":"Tstarbot-x: An open-sourced and comprehensive study for efficient league training in starcraft ii full game","author":"han","year":"2020"},{"key":"ref13","article-title":"Reinforcement learning with parameterized actions","volume":"30","author":"masson","year":"2016","journal-title":"Proceedings of the 30th AAAI Conference on Artificial Intelligence (AAAI-16)"},{"key":"ref14","article-title":"Deep reinforcement learning in parameterized action space","author":"hausknecht","year":"2015","journal-title":"Proceedings of the 4th International Conference on Learning Representations (ICLR-2016)"},{"key":"ref15","article-title":"Parametrized deep q-networks learning: Reinforcement learning with discrete-continuous hybrid action space","author":"xiong","year":"2018","journal-title":"Proceedings of the 6th International Conference on Learning Representations (ICLR-2018)"},{"key":"ref16","article-title":"Multi-pass q-networks for deep reinforcement learning with parameterised action spaces","author":"bester","year":"2019"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/323"},{"key":"ref18","article-title":"Parallel curriculum experience replay in distributed reinforcement learning","author":"li","year":"2021","journal-title":"Proceedings of the 20th International Conference on Autonomous Agents and MultiAgent Systems (AAMAS-21)"},{"key":"ref19","article-title":"Multi-agent hierarchical policy gradient for air combat tactics emergence via self-play","volume":"98","author":"piao","year":"2020","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"ref28","article-title":"Helios2018: Robocup 2018 soccer simulation 2d league champion","author":"akiyama","year":"2018"},{"key":"ref4","article-title":"Dota 2 with large scale deep reinforcement learning","author":"berner","year":"2019"},{"key":"ref27","article-title":"Alphastar: Mastering the real-time strategy game starcraft ii","volume":"2","author":"vinyals","year":"2019","journal-title":"DeepMind blog"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref6","article-title":"Tleague: A framework for competitive self-play based distributed multi-agent reinforcement learning","author":"sun","year":"2020"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICTAI50040.2020.00088"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1006\/game.1996.0065"},{"key":"ref7","article-title":"Consistency and cautious fictitious play","author":"fudenberg","year":"1996","journal-title":"Levines Working Paper Archive"},{"key":"ref2","article-title":"Half field offense: An environment for multiagent learning and ad hoc teamwork","author":"hausknecht","year":"2016","journal-title":"Proceedings of the 15th AAMAS Adaptive Learning Agents Workshop (ALA-16)"},{"key":"ref9","article-title":"Deep reinforcement learning from self-play in imperfect-information games","author":"heinrich","year":"2016"},{"key":"ref1","first-page":"62","article-title":"The robocup synthetic agent challenge 97","author":"kitano","year":"1997","journal-title":"Robot Soccer World Cup"},{"key":"ref20","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017","journal-title":"CoRR abs\/1707 06347 (2017)"},{"key":"ref22","article-title":"Massively parallel methods for deep reinforcement learning","author":"nair","year":"2015"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/316"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459670"},{"key":"ref26","article-title":"Distributed distributional deterministic policy gradients","author":"barth-maron","year":"2018","journal-title":"Proceedings of the 6th International Conference on Learning Representations (ICLR-2018)"},{"key":"ref25","article-title":"Distributed prioritized experience replay","author":"horgan","year":"2018","journal-title":"Proceedings of the 6th International Conference on Learning Representations (ICLR-2018)"}],"event":{"name":"2021 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","location":"Melbourne, Australia","start":{"date-parts":[[2021,10,17]]},"end":{"date-parts":[[2021,10,20]]}},"container-title":["2021 IEEE International Conference on Systems, Man, and Cybernetics (SMC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9658572\/9658575\/09659219.pdf?arnumber=9659219","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T16:56:33Z","timestamp":1652201793000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9659219\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/smc52423.2021.9659219","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]}}}