{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:16:38Z","timestamp":1750220198040,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,11,7]],"date-time":"2022-11-07T00:00:00Z","timestamp":1667779200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,11,7]]},"DOI":"10.1145\/3528535.3565249","type":"proceedings-article","created":{"date-parts":[[2022,12,20]],"date-time":"2022-12-20T13:40:01Z","timestamp":1671543601000},"page":"255-268","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Optimizing communication in deep reinforcement learning with\n            <i>XingTian<\/i>"],"prefix":"10.1145","author":[{"given":"Lichen","family":"Pan","sequence":"first","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Qian","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Xia","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hangyu","family":"Mao","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Yao","sequence":"additional","affiliation":[{"name":"Huawei, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengze","family":"Li","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Xiao","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,11,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 12th USENIX Symposium on Operating Systems Design and Implementation","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, Manjunath Kudlur, Josh Levenberg, Rajat Monga, Sherry Moore, Derek Gordon Murray, Benoit Steiner, Paul A. Tucker, Vijay Vasudevan, Pete Warden, Martin Wicke, Yuan Yu, and Xiaoqiang Zheng. 2016. TensorFlow: A System for Large-Scale Machine Learning. In Proceedings of the 12th USENIX Symposium on Operating Systems Design and Implementation (Savannah, GA, USA, November 2--4) (OSDI 2016). USENIX Association, USA, 265--283. https:\/\/www.usenix.org\/conference\/osdi16\/technical-sessions\/presentation\/abadi"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919887447"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the 5th International Conference on Learning Representations","author":"Babaeizadeh Mohammad","year":"2017","unstructured":"Mohammad Babaeizadeh, Iuri Frosio, Stephen Tyree, Jason Clemons, and Jan Kautz. 2017. Reinforcement Learning through Asynchronous Advantage Actor-Critic on a GPU. In Proceedings of the 5th International Conference on Learning Representations (Toulon, France, April 24--26, 2017) (ICLR 2017). OpenReview.net, USA. https:\/\/openreview.net\/forum?id=r1VGvBcxl"},{"key":"e_1_3_2_1_4_1","first-page":"679","article-title":"A Markovian decision process","volume":"6","author":"Bellman Richard","year":"1957","unstructured":"Richard Bellman. 1957. A Markovian decision process. Journal of mathematics and mechanics 6, 5 (1957), 679--684.","journal-title":"Journal of mathematics and mechanics"},{"key":"e_1_3_2_1_5_1","volume-title":"Dynamic programming and optimal control","author":"Bertsekas Dimitri P","year":"2011","unstructured":"Dimitri P Bertsekas. 2011. Dynamic programming and optimal control 3rd edition, volume II. Belmont, MA: Athena Scientific (2011).","edition":"3"},{"key":"e_1_3_2_1_6_1","volume-title":"Openai gym. arXiv preprint arXiv:1606.01540","author":"Brockman Greg","year":"2016","unstructured":"Greg Brockman, Vicki Cheung, Ludwig Pettersson, Jonas Schneider, John Schulman, Jie Tang, and Wojciech Zaremba. 2016. Openai gym. arXiv preprint arXiv:1606.01540 (2016)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.1134899"},{"key":"e_1_3_2_1_8_1","volume-title":"Reverb: A Framework For Experience Replay. arXiv preprint arXiv:2102.04736","author":"Cassirer Albin","year":"2021","unstructured":"Albin Cassirer, Gabriel Barth-Maron, Eugene Brevdo, Sabela Ramos, Toby Boyd, Thibault Sottiaux, and Manuel Kroiss. 2021. Reverb: A Framework For Experience Replay. arXiv preprint arXiv:2102.04736 (2021)."},{"key":"e_1_3_2_1_9_1","volume-title":"Dopamine: A research framework for deep reinforcement learning. arXiv preprint arXiv:1812.06110","author":"Castro Pablo Samuel","year":"2018","unstructured":"Pablo Samuel Castro, Subhodeep Moitra, Carles Gelada, Saurabh Kumar, and Marc G Bellemare. 2018. Dopamine: A research framework for deep reinforcement learning. arXiv preprint arXiv:1812.06110 (2018)."},{"key":"e_1_3_2_1_10_1","volume-title":"Efficient parallel methods for deep reinforcement learning. arXiv preprint arXiv:1705.04862","author":"Clemente Alfredo V","year":"2017","unstructured":"Alfredo V Clemente, Humberto N Castej\u00f3n, and Arjun Chandra. 2017. Efficient parallel methods for deep reinforcement learning. arXiv preprint arXiv:1705.04862 (2017)."},{"key":"e_1_3_2_1_11_1","volume-title":"Retrieved","author":"Dhariwal Prafulla","year":"2017","unstructured":"Prafulla Dhariwal, Christopher Hesse, Oleg Klimov, Alex Nichol, Matthias Plappert, Alec Radford, John Schulman, Szymon Sidor, Yuhuai Wu, and Peter Zhokhov. 2017. OpenAI Baselines. Retrieved May 15, 2022 from https:\/\/github.com\/openai\/baselines"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 8th International Conference on Learning Representations (Addis Ababa, Ethiopia, April 26--30","author":"Espeholt Lasse","year":"2020","unstructured":"Lasse Espeholt, Rapha\u00ebl Marinier, Piotr Stanczyk, Ke Wang, and Marcin Michalski. 2020. SEED RL: Scalable and Efficient Deep-RL with Accelerated Central Inference. In Proceedings of the 8th International Conference on Learning Representations (Addis Ababa, Ethiopia, April 26--30, 2020) (ICLR 2020,). OpenReview.net, USA. https:\/\/openreview.net\/forum?id=rkgvXlrKwH"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning (Stockholmsm\u00e4ssan","volume":"1415","author":"Espeholt Lasse","year":"2018","unstructured":"Lasse Espeholt, Hubert Soyer, R\u00e9mi Munos, Karen Simonyan, Volodymyr Mnih, Tom Ward, Yotam Doron, Vlad Firoiu, Tim Harley, Iain Dunning, Shane Legg, and Koray Kavukcuoglu. 2018. IMPALA: Scalable Distributed Deep-RL with Importance Weighted Actor-Learner Architectures. In Proceedings of the 35th International Conference on Machine Learning (Stockholmsm\u00e4ssan, Stockholm, Sweden, July 10--15, 2018) (ICML 2018, Vol. 80). PMLR, USA, 1406--1415. http:\/\/proceedings.mlr.press\/v80\/espeholt18a.html"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 2nd Annual Conference on Robot Learning","volume":"782","author":"Fan Linxi","year":"2018","unstructured":"Linxi Fan, Yuke Zhu, Jiren Zhu, Zihua Liu, Orien Zeng, Anchit Gupta, Joan Creus-Costa, Silvio Savarese, and Li Fei-Fei. 2018. SURREAL: Open-Source Reinforcement Learning Framework and Robot Manipulation Benchmark. In Proceedings of the 2nd Annual Conference on Robot Learning (Z\u00fcrich, Switzerland, 29--31 October 2018) (CoRL 2018, Vol. 87). PMLR, USA, 767--782. http:\/\/proceedings.mlr.press\/v87\/fan18a.html"},{"key":"e_1_3_2_1_15_1","volume-title":"Retrieved","author":"The","year":"2019","unstructured":"The garage contributors. 2019. Garage: A toolkit for reproducible reinforcement learning research. Retrieved May 13, 2022 from https:\/\/github.com\/rlworkgroup\/garage"},{"key":"e_1_3_2_1_16_1","volume-title":"Horizon: Facebook's open source applied reinforcement learning platform. arXiv preprint arXiv:1811.00260","author":"Gauci Jason","year":"2018","unstructured":"Jason Gauci, Edoardo Conti, Yitao Liang, Kittipat Virochsiri, Yuchen He, Zachary Kaden, Vivek Narayanan, Xiaohui Ye, Zhengxing Chen, and Scott Fujimoto. 2018. Horizon: Facebook's open source applied reinforcement learning platform. arXiv preprint arXiv:1811.00260 (2018)."},{"key":"e_1_3_2_1_17_1","volume-title":"Tensorflow agents: Efficient batched reinforcement learning in tensorflow. arXiv preprint arXiv:1709.02878","author":"Hafner Danijar","year":"2017","unstructured":"Danijar Hafner, James Davidson, and Vincent Vanhoucke. 2017. Tensorflow agents: Efficient batched reinforcement learning in tensorflow. arXiv preprint arXiv:1709.02878 (2017)."},{"key":"e_1_3_2_1_18_1","unstructured":"Nicolas Heess Dhruva TB Srinivasan Sriram Jay Lemmon Josh Merel Greg Wayne Yuval Tassa Tom Erez Ziyu Wang SM Eslami et al. 2017. Emergence of locomotion behaviours in rich environments. arXiv preprint arXiv:1707.02286 (2017)."},{"key":"e_1_3_2_1_19_1","volume-title":"Mohammad Gheshlaghi Azar, and David Silver","author":"Hessel Matteo","year":"2018","unstructured":"Matteo Hessel, Joseph Modayil, Hado van Hasselt, Tom Schaul, Georg Ostrovski, Will Dabney, Dan Horgan, Bilal Piot, Mohammad Gheshlaghi Azar, and David Silver. 2018. Rainbow: Combining Improvements in Deep Reinforcement Learning. In Proceedings of the Thirty-Second AAAI Conference on Artificial Intelligence (New Orleans, Louisiana, USA, February 2--7, 2018) (AAAI-18). AAAI Press, USA, 3215--3222. https:\/\/www.aaai.org\/ocs\/index.php\/AAAI\/AAAI18\/paper\/view\/17204"},{"key":"e_1_3_2_1_20_1","volume-title":"Serkan Cabi, Caglar Gulcehre, Tom Le Paine, Andrew Cowie, Ziyu Wang, Bilal Piot, and Nando de Freitas.","author":"Hoffman Matt","year":"2020","unstructured":"Matt Hoffman, Bobak Shahriari, John Aslanides, Gabriel Barth-Maron, Feryal Behbahani, Tamara Norman, Abbas Abdolmaleki, Albin Cassirer, Fan Yang, Kate Baumli, Sarah Henderson, Alex Novikov, Sergio G\u00f3mez Colmenarejo, Serkan Cabi, Caglar Gulcehre, Tom Le Paine, Andrew Cowie, Ziyu Wang, Bilal Piot, and Nando de Freitas. 2020. Acme: A Research Framework for Distributed Reinforcement Learning. arXiv preprint arXiv:2006.00979 (2020). https:\/\/arxiv.org\/abs\/2006.00979"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the 6th International Conference on Learning Representations","author":"Horgan Dan","year":"2018","unstructured":"Dan Horgan, John Quan, David Budden, Gabriel Barth-Maron, Matteo Hessel, Hado van Hasselt, and David Silver. 2018. Distributed Prioritized Experience Replay. In Proceedings of the 6th International Conference on Learning Representations (Vancouver, BC, Canada, April 30 - May 3, 2018) (ICLR 2018). OpenReview.net, USA. https:\/\/openreview.net\/forum?id=H1Dy-0Z"},{"volume-title":"Retrieved","year":"2020","key":"e_1_3_2_1_22_1","unstructured":"Huawei. 2020. MindSpore. Retrieved May 13, 2022 from https:\/\/www.mindspore.cn\/"},{"key":"e_1_3_2_1_23_1","volume-title":"Charles Beattie, Neil C Rabinowitz, Ari S Morcos, Avraham Ruderman, et al.","author":"Jaderberg Max","year":"2019","unstructured":"Max Jaderberg, Wojciech M Czarnecki, Iain Dunning, Luke Marris, Guy Lever, Antonio Garcia Castaneda, Charles Beattie, Neil C Rabinowitz, Ari S Morcos, Avraham Ruderman, et al. 2019. Human-level performance in 3D multiplayer games with population-based reinforcement learning. Science 364, 6443 (2019), 859--865."},{"key":"e_1_3_2_1_24_1","volume-title":"Highly accurate protein structure prediction with AlphaFold. Nature 596, 7873 (01","author":"Jumper John","year":"2021","unstructured":"John Jumper, Richard Evans, Alexander Pritzel, Tim Green, Michael Figurnov, Olaf Ronneberger, Kathryn Tunyasuvunakool, Russ Bates, Augustin \u017d\u00eddek, Anna Potapenko, Alex Bridgland, Clemens Meyer, Simon A. A. Kohl, Andrew J. Ballard, Andrew Cowie, Bernardino Romera-Paredes, Stanislav Nikolov, Rishub Jain, Jonas Adler, Trevor Back, Stig Petersen, David Reiman, Ellen Clancy, Michal Zielinski, Martin Steinegger, Michalina Pacholska, Tamas Berghammer, Sebastian Bodenstein, David Silver, Oriol Vinyals, Andrew W. Senior, Koray Kavukcuoglu, Pushmeet Kohli, and Demis Hassabis. 2021. Highly accurate protein structure prediction with AlphaFold. Nature 596, 7873 (01 Aug 2021), 583--589."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 8th International Conference on Learning Representations (Addis Ababa, Ethiopia, April 26--30","author":"Kaiser Lukasz","year":"2020","unstructured":"Lukasz Kaiser, Mohammad Babaeizadeh, Piotr Milos, Blazej Osinski, Roy H. Campbell, Konrad Czechowski, Dumitru Erhan, Chelsea Finn, Piotr Kozakowski, Sergey Levine, Afroz Mohiuddin, Ryan Sepassi, George Tucker, and Henryk Michalewski. 2020. Model Based Reinforcement Learning for Atari. In Proceedings of the 8th International Conference on Learning Representations (Addis Ababa, Ethiopia, April 26--30, 2020) (ICLR 2020). OpenReview.net, USA. https:\/\/openreview.net\/forum?id=S1xCPJHtDB"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 7th International Conference on Learning Representations","author":"Kapturowski Steven","year":"2019","unstructured":"Steven Kapturowski, Georg Ostrovski, John Quan, Remi Munos, and Will Dabney. 2019. Recurrent Experience Replay in Distributed Reinforcement Learning. In Proceedings of the 7th International Conference on Learning Representations (New Orleans, LA, USA, May 6--9, 2019) (ICLR 2019). OpenReview.net, USA. https:\/\/openreview.net\/forum?id=r1lyTjAqYX"},{"key":"e_1_3_2_1_27_1","volume-title":"Tsitsiklis","author":"Konda Vijay R.","year":"1999","unstructured":"Vijay R. Konda and John N. Tsitsiklis. 1999. Actor-Critic Algorithms. In Proceedings of the Advances in Neural Information Processing Systems (Denver, Colorado, USA, November 29 - December 4, 1999) (NIPS Conference 1999). The MIT Press, USA, 1008--1014. http:\/\/papers.nips.cc\/paper\/1786-actor-critic-algorithms"},{"key":"e_1_3_2_1_28_1","volume-title":"Retrieved","author":"Kuhnle Alexander","year":"2017","unstructured":"Alexander Kuhnle, Michael Schaarschmidt, and Kai Fricke. 2017. Tensorforce: a TensorFlow library for applied reinforcement learning. Retrieved May 14, 2022 from https:\/\/github.com\/tensorforce\/tensorforce"},{"key":"e_1_3_2_1_29_1","volume-title":"Torchbeast: A pytorch platform for distributed rl. arXiv preprint arXiv:1910.03552","author":"K\u00fcttler Heinrich","year":"2019","unstructured":"Heinrich K\u00fcttler, Nantas Nardelli, Thibaut Lavril, Marco Selvatici, Viswanath Sivakumar, Tim Rockt\u00e4schel, and Edward Grefenstette. 2019. Torchbeast: A pytorch platform for distributed rl. arXiv preprint arXiv:1910.03552 (2019)."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning (Stockholmsm\u00e4ssan","volume":"3068","author":"Liang Eric","year":"2018","unstructured":"Eric Liang, Richard Liaw, Robert Nishihara, Philipp Moritz, Roy Fox, Ken Goldberg, Joseph Gonzalez, Michael I. Jordan, and Ion Stoica. 2018. RLlib: Abstractions for Distributed Reinforcement Learning. In Proceedings of the 35th International Conference on Machine Learning (Stockholmsm\u00e4ssan, Stockholm, Sweden, July 10--15, 2018) (ICML 2018, Vol. 80). PMLR, USA, 3059--3068. http:\/\/proceedings.mlr.press\/v80\/liang18b.html"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021 (Virtual Event, December 6--14","author":"Liang Eric","year":"2021","unstructured":"Eric Liang, Zhanghao Wu, Michael Luo, Sven Mika, Joseph E. Gonzalez, and Ion Stoica. 2021. RLlib Flow: Distributed Reinforcement Learning is a Dataflow Problem. In Proceedings of the Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021 (Virtual Event, December 6--14, 2021) (NeurIPS 2021). USA, 5506--5517. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/2bce32ed409f5ebcee2a7b417ad9beed-Abstract.html"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 4th International Conference on Learning Representations","author":"Lillicrap Timothy P.","year":"2016","unstructured":"Timothy P. Lillicrap, Jonathan J. Hunt, Alexander Pritzel, Nicolas Heess, Tom Erez, Yuval Tassa, David Silver, and Daan Wierstra. 2016. Continuous control with deep reinforcement learning. In Proceedings of the 4th International Conference on Learning Representations (San Juan, Puerto Rico, May 2--4, 2016) (ICLR 2016). USA. http:\/\/arxiv.org\/abs\/1509.02971"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8461113"},{"key":"e_1_3_2_1_34_1","volume-title":"Deep Reinforcement Learning Versus Evolution Strategies: A Comparative Survey. arXiv preprint arXiv:2110.01411","author":"Majid Amjad Yousef","year":"2021","unstructured":"Amjad Yousef Majid, Serge Saaybi, Tomas van Rietbergen, Vincent Francois-Lavet, R Venkatesha Prasad, and Chris Verhoeven. 2021. Deep Reinforcement Learning Versus Evolution Strategies: A Comparative Survey. arXiv preprint arXiv:2110.01411 (2021)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6212"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5957"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the 33nd International Conference on Machine Learning","volume":"1937","author":"Mnih Volodymyr","year":"2016","unstructured":"Volodymyr Mnih, Adri\u00e0 Puigdom\u00e8nech Badia, Mehdi Mirza, Alex Graves, Timothy P. Lillicrap, Tim Harley, David Silver, and Koray Kavukcuoglu. 2016. Asynchronous Methods for Deep Reinforcement Learning. In Proceedings of the 33nd International Conference on Machine Learning (New York City, NY, USA, June 19--24, 2016) (ICML 2016, Vol. 48). JMLR.org, USA, 1928--1937. http:\/\/proceedings.mlr.press\/v48\/mniha16.html"},{"key":"e_1_3_2_1_38_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller. 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the 13th USENIX Conference on Operating Systems Design and Implementation","author":"Moritz Philipp","year":"2018","unstructured":"Philipp Moritz, Robert Nishihara, Stephanie Wang, Alexey Tumanov, Richard Liaw, Eric Liang, Melih Elibol, Zongheng Yang, William Paul, Michael I. Jordan, and Ion Stoica. 2018. Ray: A Distributed Framework for Emerging AI Applications. In Proceedings of the 13th USENIX Conference on Operating Systems Design and Implementation (Carlsbad, CA, USA, October 8--10, 2018) (OSDI 2018). USENIX Association, USA, 561--577."},{"key":"e_1_3_2_1_40_1","volume-title":"Vedavyas Panneershelvam, Mustafa Suleyman, Charles Beattie, Stig Petersen, et al.","author":"Nair Arun","year":"2015","unstructured":"Arun Nair, Praveen Srinivasan, Sam Blackwell, Cagdas Alcicek, Rory Fearon, Alessandro De Maria, Vedavyas Panneershelvam, Mustafa Suleyman, Charles Beattie, Stig Petersen, et al. 2015. Massively parallel methods for deep reinforcement learning. arXiv preprint arXiv:1507.04296 (2015)."},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Z. Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In Proceedings of the Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019 (Vancouver, BC, Canada, December 8--14, 2019) (NeurIPS 2019). USA, 8024--8035. https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/bdbca288fee7f92f2bfa9f7012727740-Abstract.html"},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Virtual Event, 13--18","volume":"7662","author":"Petrenko Aleksei","year":"2020","unstructured":"Aleksei Petrenko, Zhehui Huang, Tushar Kumar, Gaurav S. Sukhatme, and Vladlen Koltun. 2020. Sample Factory: Egocentric 3D Control from Pixels at 100000 FPS with Asynchronous Reinforcement Learning. In Proceedings of the 37th International Conference on Machine Learning (Virtual Event, 13--18 July 2020) (ICML 2020, Vol. 119). PMLR, USA, 7652--7662. http:\/\/proceedings.mlr.press\/v119\/petrenko20a.html"},{"key":"e_1_3_2_1_43_1","volume-title":"Retrieved","author":"Plappert Matthias","year":"2016","unstructured":"Matthias Plappert. 2016. keras-rl. Retrieved May 13, 2022 from https:\/\/github.com\/keras-rl\/keras-rl"},{"key":"e_1_3_2_1_44_1","volume-title":"Retrieved","author":"Raffin Antonin","year":"2019","unstructured":"Antonin Raffin, Ashley Hill, Maximilian Ernestus, Adam Gleave, Anssi Kanervisto, and Noah Dormann. 2019. Stable Baselines3. Retrieved May 13, 2022 from https:\/\/github.com\/DLR-RM\/stable-baselines3"},{"volume-title":"On-line Q-learning using connectionist systems","author":"Rummery Gavin A","key":"e_1_3_2_1_45_1","unstructured":"Gavin A Rummery and Mahesan Niranjan. 1994. On-line Q-learning using connectionist systems. Vol. 37. Citeseer."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of Machine Learning and Systems 2019","author":"Schaarschmidt Michael","year":"2019","unstructured":"Michael Schaarschmidt, Sven Mika, Kai Fricke, and Eiko Yoneki. 2019. RLgraph: Modular Computation Graphs for Deep Reinforcement Learning. In Proceedings of Machine Learning and Systems 2019 (Stanford, CA, USA, March 31 - April 2, 2019) (Mlsys 2019). mlsys.org, USA, 65--80. https:\/\/proceedings.mlsys.org\/book\/279.pdf"},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 4th International Conference on Learning Representations","author":"Schaul Tom","year":"2016","unstructured":"Tom Schaul, John Quan, Ioannis Antonoglou, and David Silver. 2016. Prioritized Experience Replay. In Proceedings of the 4th International Conference on Learning Representations (San Juan, Puerto Rico, May 2--4, 2016) (ICLR 2016). USA. http:\/\/arxiv.org\/abs\/1511.05952"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Julian Schrittwieser Ioannis Antonoglou Thomas Hubert Karen Simonyan Laurent Sifre Simon Schmitt Arthur Guez Edward Lockhart Demis Hassabis Thore Graepel et al. 2020. Mastering atari go chess and shogi by planning with a learned model. Nature 588 7839 (2020) 604--609.","DOI":"10.1038\/s41586-020-03051-4"},{"key":"e_1_3_2_1_49_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_50_1","volume-title":"Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et al.","author":"Silver David","year":"2016","unstructured":"David Silver, Aja Huang, Chris J Maddison, Arthur Guez, Laurent Sifre, George Van Den Driessche, Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et al. 2016. Mastering the game of Go with deep neural networks and tree search. nature 529, 7587 (2016), 484--489."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"David Silver Thomas Hubert Julian Schrittwieser Ioannis Antonoglou Matthew Lai Arthur Guez Marc Lanctot Laurent Sifre Dharshan Kumaran Thore Graepel et al. 2018. A general reinforcement learning algorithm that masters chess shogi and Go through self-play. Science 362 6419 (2018) 1140--1144.","DOI":"10.1126\/science.aar6404"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"crossref","unstructured":"David Silver Julian Schrittwieser Karen Simonyan Ioannis Antonoglou Aja Huang Arthur Guez Thomas Hubert Lucas Baker Matthew Lai Adrian Bolton et al. 2017. Mastering the game of go without human knowledge. nature 550 7676 (2017) 354--359.","DOI":"10.1038\/nature24270"},{"key":"e_1_3_2_1_53_1","volume-title":"Accelerated methods for deep reinforcement learning. arXiv preprint arXiv:1803.02811","author":"Stooke Adam","year":"2018","unstructured":"Adam Stooke and Pieter Abbeel. 2018. Accelerated methods for deep reinforcement learning. arXiv preprint arXiv:1803.02811 (2018)."},{"key":"e_1_3_2_1_54_1","volume-title":"rlpyt: A research code base for deep reinforcement learning in pytorch. arXiv preprint arXiv:1909.01500","author":"Stooke Adam","year":"2019","unstructured":"Adam Stooke and Pieter Abbeel. 2019. rlpyt: A research code base for deep reinforcement learning in pytorch. arXiv preprint arXiv:1909.01500 (2019)."},{"volume-title":"Reinforcement learning: An introduction","author":"Sutton Richard S","key":"e_1_3_2_1_55_1","unstructured":"Richard S Sutton and Andrew G Barto. 2018. Reinforcement learning: An introduction. MIT press."},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (Denver, Colorado, USA, November 29 -","author":"Sutton Richard S.","year":"1999","unstructured":"Richard S. Sutton, David A. McAllester, Satinder Singh, and Yishay Mansour. 1999. Policy Gradient Methods for Reinforcement Learning with Function Approximation. In Proceedings of the Advances in Neural Information Processing Systems (Denver, Colorado, USA, November 29 - December 4, 1999) (NIPS Conference 1999). The MIT Press, USA, 1057--1063. http:\/\/papers.nips.cc\/paper\/1713-policy-gradient-methods-for-reinforcement-learning-with-function-approximation"},{"key":"e_1_3_2_1_57_1","volume-title":"Retrieved","author":"Development Team Apache Arrow","year":"2021","unstructured":"Apache Arrow Development Team. 2021. Apache Arrow. Retrieved May 13, 2022 from https:\/\/arrow.apache.org\/"},{"key":"e_1_3_2_1_58_1","first-page":"1","article-title":"Tianshou: A Highly Modularized Deep Reinforcement Learning Library","volume":"23","author":"Weng Jiayi","year":"2022","unstructured":"Jiayi Weng, Huayu Chen, Dong Yan, Kaichao You, Alexis Duburcq, Minghao Zhang, Yi Su, Hang Su, and Jun Zhu. 2022. Tianshou: A Highly Modularized Deep Reinforcement Learning Library. Journal of Machine Learning Research 23, 267 (2022), 1--6. http:\/\/jmlr.org\/papers\/v23\/21-1127.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_59_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning 8, 3","author":"Williams Ronald J","year":"1992","unstructured":"Ronald J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning 8, 3 (1992), 229--256."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/80"},{"key":"e_1_3_2_1_61_1","volume-title":"Launchpad: A Programming Model for Distributed Machine Learning Research. arXiv preprint arXiv:2106.04516","author":"Yang Fan","year":"2021","unstructured":"Fan Yang, Gabriel Barth-Maron, Piotr Sta\u0144czyk, Matthew Hoffman, Siqi Liu, Manuel Kroiss, Aedan Pope, and Alban Rrustemi. 2021. Launchpad: A Programming Model for Distributed Machine Learning Research. arXiv preprint arXiv:2106.04516 (2021)."},{"key":"e_1_3_2_1_62_1","volume-title":"Fiber: A platform for efficient development and distributed training for reinforcement learning and population-based methods. arXiv preprint arXiv:2003.11164","author":"Zhi Jiale","year":"2020","unstructured":"Jiale Zhi, Rui Wang, Jeff Clune, and Kenneth O Stanley. 2020. Fiber: A platform for efficient development and distributed training for reinforcement learning and population-based methods. arXiv preprint arXiv:2003.11164 (2020)."}],"event":{"name":"Middleware '22: 23rd International Middleware Conference","sponsor":["ACM Association for Computing Machinery","IFIP"],"location":"Quebec QC Canada","acronym":"Middleware '22"},"container-title":["Proceedings of the 23rd ACM\/IFIP International Middleware Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3528535.3565249","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3528535.3565249","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:43Z","timestamp":1750186963000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3528535.3565249"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,7]]},"references-count":62,"alternative-id":["10.1145\/3528535.3565249","10.1145\/3528535"],"URL":"https:\/\/doi.org\/10.1145\/3528535.3565249","relation":{},"subject":[],"published":{"date-parts":[[2022,11,7]]},"assertion":[{"value":"2022-11-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}