{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,22]],"date-time":"2024-10-22T22:17:17Z","timestamp":1729635437192,"version":"3.28.0"},"reference-count":76,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,9,22]],"date-time":"2020-09-22T00:00:00Z","timestamp":1600732800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,9,22]]},"DOI":"10.1109\/hpec43674.2020.9286212","type":"proceedings-article","created":{"date-parts":[[2020,12,22]],"date-time":"2020-12-22T21:07:15Z","timestamp":1608671235000},"page":"1-9","source":"Crossref","is-referenced-by-count":0,"title":["Towards a Distributed Framework for Multi-Agent Reinforcement Learning Research"],"prefix":"10.1109","author":[{"given":"Yutai","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shawn","family":"Manuel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peter","family":"Morales","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jaime","family":"Pena","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ross","family":"Allen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","article-title":"Credit assignment for collective multiagent rl with global rewards","author":"nguyen","year":"2018","journal-title":"NeurIPS"},{"key":"ref72","volume":"abs 1902 7151","author":"liu","year":"2019","journal-title":"Emergent coordination through competition"},{"key":"ref71","first-page":"44","article-title":"Slurm: Simple linux utility for resource management","author":"jette","year":"2002","journal-title":"In Lecture Notes in Computer Science Proceedings of Job Scheduling Strategies for Parallel Processing (JSSPP) 2003"},{"key":"ref70","volume":"abs 1604 6778","author":"duan","year":"2016","journal-title":"Benchmarking deep reinforcement learning for continuous control"},{"key":"ref76","article-title":"Learning when to communicate at scale in multiagent cooperative and competitive tasks","author":"singh","year":"2019","journal-title":"7th International Conference on Learning Representations ICLR 2019"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1145\/544741.544832"},{"journal-title":"Deep multi-agent reinforcement learning for decentralized continuous cooperative control","year":"2020","author":"de witt","key":"ref39"},{"journal-title":"Deep implicit coordination graphs for multi-agent reinforcement learning","year":"2020","author":"li","key":"ref75"},{"key":"ref38","doi-asserted-by":"crossref","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","article-title":"Grandmaster level in starcraft ii using multiagent reinforcement learning","volume":"575","author":"vinyals","year":"2019","journal-title":"Nature"},{"key":"ref33","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","author":"lowe","year":"2017","journal-title":"Neural Information Processing Systems (NIPS)"},{"journal-title":"Proximal policy optimization algorithms","year":"2017","author":"schulman","key":"ref32"},{"key":"ref31","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref30","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"International Conference on Machine Learning"},{"journal-title":"Starcraft ii A new challenge for reinforcement learning","year":"2017","author":"vinyals","key":"ref37"},{"journal-title":"Unity A General Platform for Intelligent Agents","year":"2018","author":"juliani","key":"ref36"},{"journal-title":"OpenAI Gym","year":"2016","author":"brockman","key":"ref35"},{"key":"ref34","doi-asserted-by":"crossref","DOI":"10.1609\/aaai.v32i1.11794","article-title":"Counterfactual multi-agent policy gradients","author":"foerster","year":"2018","journal-title":"Thirty-Second AAAI Conference on Artificial Intelligence"},{"journal-title":"OpenAI Baselines","year":"2017","author":"dhariwal","key":"ref60"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo. 1134899"},{"journal-title":"Stable Baselines","year":"2018","author":"hill","key":"ref61"},{"journal-title":"Tensorforce A tensorflow library for applied reinforcement learning","year":"2017","author":"kuhnle","key":"ref63"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"229","DOI":"10.1007\/BF00992696","article-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning","volume":"8","author":"williams","year":"1992","journal-title":"Machine Learning"},{"journal-title":"TF-Agents A library for reinforcement learning in tensorflow","year":"2018","author":"guadarrama","key":"ref64"},{"key":"ref27","first-page":"20","article-title":"Exact dynamic programming for decentralized pomdps with lossless policy compression","author":"boularias","year":"2008","journal-title":"Eighteenth International Conference on Automated Planning and Scheduling (ICAPS'08)"},{"key":"ref65","volume":"abs 1909 12989","author":"fan","year":"2019","journal-title":"Surreal-system Fully-integrated stack for distributed deep reinforcement learning"},{"journal-title":"Arena A general evaluation platform and building toolkit for multiagent intelligence","year":"2019","author":"song","key":"ref66"},{"journal-title":"Continuous control with deep reinforcement learning","year":"2015","author":"lillicrap","key":"ref29"},{"key":"ref67","volume":"abs 1810 9028","author":"schaarschmidt","year":"2018","journal-title":"R1-graph Flexible computation graphs for deep reinforcement learning"},{"journal-title":"Acme A research framework for distributed reinforcement learning","year":"2020","author":"hoffman","key":"ref68"},{"key":"ref69","volume":"abs 1909 1500","author":"stooke","year":"2019","journal-title":"Rlpyt A research code base for deep reinforcement learning in pytorch"},{"journal-title":"Reinforcement Learning An Introduction","year":"2018","author":"sutton","key":"ref2"},{"journal-title":"Distributed Systems Principles and Paradigms","year":"2006","author":"tanenbaum","key":"ref1"},{"journal-title":"Google research football A novel reinforcement learning environment","year":"2019","author":"kurach","key":"ref20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref21","first-page":"2186","article-title":"The starcraft multi -agent challenge","author":"samvelyan","year":"2019","journal-title":"Int Conf Auton Agents Multiagent syst International Foundation for Autonomous Agents and Multiagent Systems"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/10187.001.0001"},{"journal-title":"Deep multi-agent reinforcement learning for decentralized continuous cooperative control","year":"2020","author":"de witt","key":"ref23"},{"journal-title":"High-dimensional continuous control using generalized advantage estimation","year":"2015","author":"schulman","key":"ref26"},{"key":"ref25","first-page":"709","article-title":"Dynamic programming for partially observable stochastic games","volume":"4","author":"hansen","year":"2004","journal-title":"AAAI"},{"journal-title":"Assessing generalization in deep reinforcement learning","year":"2018","author":"packer","key":"ref50"},{"key":"ref51","article-title":"The StarCraft Multi-Agent Challenge","volume":"abs 1902 4043","author":"samvelyan","year":"2019","journal-title":"CoRR"},{"journal-title":"Reinforcement Learning An Introduction","year":"2018","author":"sutton","key":"ref59"},{"journal-title":"Emergence of Grounded Compositional Language in Multi-Agent Populations","year":"2017","author":"mordatch","key":"ref58"},{"key":"ref57","article-title":"Investigating contingency awareness using atari 2600 games","author":"bellemare","year":"2012","journal-title":"AAAI"},{"key":"ref56","volume":"abs 1801 690","author":"tassa","year":"2018","journal-title":"Deepmind control suite"},{"journal-title":"Ma-gym","year":"2019","author":"koul","key":"ref55"},{"journal-title":"Is Deep Reinforcement Learning Really Superhuman on Atari? Leveling the playing field","year":"2019","author":"toromanoff","key":"ref54"},{"key":"ref53","article-title":"How many random seeds? statistical power analysis in deep reinforcement learning experiments","volume":"abs 1806 8295","author":"colas","year":"2018","journal-title":"CoRR"},{"journal-title":"Deep reinforcement learning that matters","year":"2017","author":"henderson","key":"ref52"},{"journal-title":"Are commercial labs stealing academia's ai thunder?","year":"2019","author":"peng","key":"ref10"},{"key":"ref11","article-title":"Pytorch: An imperative style, high-performance deep learning library","author":"paszke","year":"2019","journal-title":"NeurIPS"},{"key":"ref40","first-page":"1","article-title":"Introduction to discrete-event simulation and the simpy language","volume":"2","author":"matloff","year":"2008","journal-title":"Davis CA Dept of Computer Science University of California at Davis Retrieved on August"},{"key":"ref12","first-page":"265","article-title":"Tensorflow: A system for large-scale machine learning","author":"abadi","year":"2016","journal-title":"12th USENIX Symposium on Operating Systems Design and Implementation (OSDI 16)"},{"key":"ref13","article-title":"Horizon: Facebook's open source applied reinforcement learning platform","volume":"abs 1811 260","author":"gauci","year":"2018","journal-title":"CoRR"},{"key":"ref14","first-page":"3053","article-title":"RLlib: Abstractions for distributed reinforcement learning","volume":"80","author":"liang","year":"2018","journal-title":"Proceedings of the 35th International Conference on Machine Learning"},{"key":"ref15","first-page":"561","article-title":"Ray: A distributed framework for emerging ai applications","author":"moritz","year":"2018","journal-title":"Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation ser OSDI&#x2019; 18"},{"key":"ref16","article-title":"Dopamine: A research framework for deep reinforcement learning","volume":"abs 1812 6110","author":"castro","year":"2018","journal-title":"CoRR"},{"journal-title":"Garage A toolkit for reproducible reinforcement learning research","year":"2019","key":"ref17"},{"key":"ref18","article-title":"Openai gym","volume":"abs 1606 1540","author":"brockman","year":"2016","journal-title":"CoRR"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2018.8547629"},{"key":"ref4","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume":"48","author":"mnih","year":"2016","journal-title":"Proceedings of the 33rd International Conference on Machine Learning"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77949-0_5"},{"key":"ref6","first-page":"1889","article-title":"Trust region policy optimization","volume":"37","author":"schulman","year":"2015","journal-title":"Proceedings of The 32nd International Conference on Machine Learning"},{"journal-title":"Prioritized experience replay","year":"2015","author":"schaul","key":"ref5"},{"key":"ref8","first-page":"1523","article-title":"Multiagent planning with factored MDPs","author":"guestrin","year":"2002","journal-title":"Advances in Neural Information Processing Systems (Ne u rIPS)"},{"key":"ref7","article-title":"Proximal policy optimization algorithms","volume":"abs 1707 6347","author":"schulman","year":"2017","journal-title":"CoRR"},{"journal-title":"The minerl competition on sample efficient reinforcement learning using human priors","year":"2019","author":"guss","key":"ref49"},{"key":"ref9","doi-asserted-by":"crossref","first-page":"66","DOI":"10.1007\/978-3-319-71682-4_5","article-title":"Cooperative multi-agent control using deep reinforcement learning","author":"gupta","year":"2017","journal-title":"International Conference on Autonomous Agents and Multiagent Systems (AAMAS)"},{"journal-title":"Alphastar Mastering the real-time strategy game StarCraft II","year":"2019","author":"vinyals","key":"ref46"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"journal-title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","year":"2019","author":"cobbe","key":"ref48"},{"journal-title":"OpenAI OpenAI Five","year":"2018","key":"ref47"},{"key":"ref42","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2018.2848264"},{"key":"ref44","volume":"abs 1912 6680","author":"berner","year":"2019","journal-title":"Dota 2 with large scale deep reinforcement learning"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"}],"event":{"name":"2020 IEEE High Performance Extreme Computing Conference (HPEC)","start":{"date-parts":[[2020,9,22]]},"location":"Waltham, MA, USA","end":{"date-parts":[[2020,9,24]]}},"container-title":["2020 IEEE High Performance Extreme Computing Conference (HPEC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9285977\/9286137\/09286212.pdf?arnumber=9286212","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,8]],"date-time":"2022-12-08T20:36:06Z","timestamp":1670531766000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9286212\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,9,22]]},"references-count":76,"URL":"https:\/\/doi.org\/10.1109\/hpec43674.2020.9286212","relation":{},"subject":[],"published":{"date-parts":[[2020,9,22]]}}}