{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T07:48:38Z","timestamp":1767340118391,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":20,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,11,26]],"date-time":"2021-11-26T00:00:00Z","timestamp":1637884800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"This research has been co?financed by the European Regional Development Fund of the European Union and Greek national funds through the Operational Program Competitiveness, Entrepreneurship and Innovation, under the call ``Special Actions Aquaculture - Industrial Materials - Open Innovation in Culture' (project code:T6YBP-00166)."}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,11,26]]},"DOI":"10.1145\/3503823.3503905","type":"proceedings-article","created":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T22:15:51Z","timestamp":1645568151000},"page":"448-453","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Shaping the Behavior of Reinforcement Learning Agents"],"prefix":"10.1145","author":[{"given":"George","family":"Sidiropoulos","sequence":"first","affiliation":[{"name":"Athena-Research and Innovation Center in Information, Communication and Knowledge Technologies, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chairi","family":"Kiourt","sequence":"additional","affiliation":[{"name":"Athena-Research and Innovation Center in Information, Communication and Knowledge Technologies, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vasileios","family":"Sevetlidis","sequence":"additional","affiliation":[{"name":"Athena-Research and Innovation Center in Information, Communication and Knowledge Technologies, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"George","family":"Pavlidis","sequence":"additional","affiliation":[{"name":"Athena-Research and Innovation Center in Information, Communication and Knowledge Technologies, Greece"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,2,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World. 2020 International Joint Conference on Neural Networks (IJCNN) (July 2020","author":"Alm\u00e1si P\u00e9ter","year":"2020","unstructured":"P\u00e9ter Alm\u00e1si , R\u00f3bert Moni , and B\u00e1lint Gyires-T\u00f3th . 2020 . Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World. 2020 International Joint Conference on Neural Networks (IJCNN) (July 2020 ), 1\u20138. https:\/\/doi.org\/10.1109\/IJCNN48605.2020.9207497 arXiv:2009.11212. P\u00e9ter Alm\u00e1si, R\u00f3bert Moni, and B\u00e1lint Gyires-T\u00f3th. 2020. Robust Reinforcement Learning-based Autonomous Driving Agent for Simulation and Real World. 2020 International Joint Conference on Neural Networks (IJCNN) (July 2020), 1\u20138. https:\/\/doi.org\/10.1109\/IJCNN48605.2020.9207497 arXiv:2009.11212."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1907370117"},{"key":"e_1_3_2_1_3_1","volume-title":"Measuring collaborative emergent behavior in multi-agent reinforcement learning. arXiv:1807.08663 [cs] (July","author":"Barton L.","year":"2018","unstructured":"Sean\u00a0 L. Barton , Nicholas\u00a0 R. Waytowich , Erin Zaroukian , and Derrik\u00a0 E. Asher . 2018. Measuring collaborative emergent behavior in multi-agent reinforcement learning. arXiv:1807.08663 [cs] (July 2018 ). http:\/\/arxiv.org\/abs\/1807.08663 arXiv:1807.08663. Sean\u00a0L. Barton, Nicholas\u00a0R. Waytowich, Erin Zaroukian, and Derrik\u00a0E. Asher. 2018. Measuring collaborative emergent behavior in multi-agent reinforcement learning. arXiv:1807.08663 [cs] (July 2018). http:\/\/arxiv.org\/abs\/1807.08663 arXiv:1807.08663."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-015-2547-z"},{"key":"e_1_3_2_1_5_1","unstructured":"Jakob Foerster Gregory Farquhar Triantafyllos Afouras Nantas Nardelli and Shimon Whiteson. 2017. Counterfactual Multi-Agent Policy Gradients. arxiv:1705.08926\u00a0[cs.AI]  Jakob Foerster Gregory Farquhar Triantafyllos Afouras Nantas Nardelli and Shimon Whiteson. 2017. Counterfactual Multi-Agent Policy Gradients. arxiv:1705.08926\u00a0[cs.AI]"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1126\/science.aau6249"},{"key":"e_1_3_2_1_7_1","volume-title":"Unity: A General Platform for Intelligent Agents. arxiv:1809.02627\u00a0[cs.LG]","author":"Juliani Arthur","year":"2020","unstructured":"Arthur Juliani , Vincent-Pierre Berges , Ervin Teng , Andrew Cohen , Jonathan Harper , Chris Elion , Chris Goy , Yuan Gao , Hunter Henry , Marwan Mattar , and Danny Lange . 2020 . Unity: A General Platform for Intelligent Agents. arxiv:1809.02627\u00a0[cs.LG] Arthur Juliani, Vincent-Pierre Berges, Ervin Teng, Andrew Cohen, Jonathan Harper, Chris Elion, Chris Goy, Yuan Gao, Hunter Henry, Marwan Mattar, and Danny Lange. 2020. Unity: A General Platform for Intelligent Agents. arxiv:1809.02627\u00a0[cs.LG]"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1177\/1059712316679239"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-018-1234-6"},{"key":"e_1_3_2_1_10_1","volume-title":"Langlois and Tom Everitt","author":"D.","year":"2021","unstructured":"Eric\u00a0 D. Langlois and Tom Everitt . 2021 . How RL Agents Behave When Their Actions Are Modified. In AAAI. Eric\u00a0D. Langlois and Tom Everitt. 2021. How RL Agents Behave When Their Actions Are Modified. In AAAI."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2010.5586191"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295385"},{"key":"e_1_3_2_1_13_1","volume-title":"Simulating Human Behavior in Fighting Games Using Reinforcement Learning and Artificial Neural Networks. In 2015 14th Brazilian Symposium on Computer Games and Digital Entertainment (SBGames). 152\u2013159","author":"Mendon\u00e7a Matheus","year":"2015","unstructured":"Matheus R.\u00a0F. Mendon\u00e7a , Heder\u00a0 S. Bernardino , and Raul\u00a0 F. Neto . 2015 . Simulating Human Behavior in Fighting Games Using Reinforcement Learning and Artificial Neural Networks. In 2015 14th Brazilian Symposium on Computer Games and Digital Entertainment (SBGames). 152\u2013159 . https:\/\/doi.org\/10.1109\/SBGames.2015.25 Matheus R.\u00a0F. Mendon\u00e7a, Heder\u00a0S. Bernardino, and Raul\u00a0F. Neto. 2015. Simulating Human Behavior in Fighting Games Using Reinforcement Learning and Artificial Neural Networks. In 2015 14th Brazilian Symposium on Computer Games and Digital Entertainment (SBGames). 152\u2013159. https:\/\/doi.org\/10.1109\/SBGames.2015.25"},{"key":"e_1_3_2_1_14_1","unstructured":"OpenAI. 2018. OpenAI Five. https:\/\/blog.openai.com\/openai-five\/.  OpenAI. 2018. OpenAI Five. https:\/\/blog.openai.com\/openai-five\/."},{"key":"e_1_3_2_1_15_1","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arxiv:1707.06347\u00a0[cs.LG]  John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arxiv:1707.06347\u00a0[cs.LG]"},{"key":"e_1_3_2_1_16_1","volume-title":"Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments. arXiv:2005.13857 [cs] (May","author":"Surmann Hartmut","year":"2020","unstructured":"Hartmut Surmann , Christian Jestel , Robin Marchel , Franziska Musberg , Houssem Elhadj , and Mahbube Ardani . 2020. Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments. arXiv:2005.13857 [cs] (May 2020 ). http:\/\/arxiv.org\/abs\/2005.13857 arXiv:2005.13857. Hartmut Surmann, Christian Jestel, Robin Marchel, Franziska Musberg, Houssem Elhadj, and Mahbube Ardani. 2020. Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments. arXiv:2005.13857 [cs] (May 2020). http:\/\/arxiv.org\/abs\/2005.13857 arXiv:2005.13857."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.5555\/3312046"},{"key":"e_1_3_2_1_18_1","volume-title":"SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II. arXiv:2012.13169 [cs] (May","author":"Wang Xiangjun","year":"2021","unstructured":"Xiangjun Wang , Junxiao Song , Penghui Qi , Peng Peng , Zhenkun Tang , Wei Zhang , Weimin Li , Xiongjun Pi , Jujie He , Chao Gao , Haitao Long , and Quan Yuan . 2021. SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II. arXiv:2012.13169 [cs] (May 2021 ). http:\/\/arxiv.org\/abs\/2012.13169 arXiv:2012.13169. Xiangjun Wang, Junxiao Song, Penghui Qi, Peng Peng, Zhenkun Tang, Wei Zhang, Weimin Li, Xiongjun Pi, Jujie He, Chao Gao, Haitao Long, and Quan Yuan. 2021. SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II. arXiv:2012.13169 [cs] (May 2021). http:\/\/arxiv.org\/abs\/2012.13169 arXiv:2012.13169."},{"key":"e_1_3_2_1_19_1","volume-title":"Robust Imitation of Diverse Behaviors. arXiv:1707.02747 [cs] (July","author":"Wang Ziyu","year":"2017","unstructured":"Ziyu Wang , Josh Merel , Scott Reed , Greg Wayne , Nando de Freitas , and Nicolas Heess . 2017. Robust Imitation of Diverse Behaviors. arXiv:1707.02747 [cs] (July 2017 ). http:\/\/arxiv.org\/abs\/1707.02747 arXiv:1707.02747. Ziyu Wang, Josh Merel, Scott Reed, Greg Wayne, Nando de Freitas, and Nicolas Heess. 2017. Robust Imitation of Diverse Behaviors. arXiv:1707.02747 [cs] (July 2017). http:\/\/arxiv.org\/abs\/1707.02747 arXiv:1707.02747."},{"key":"e_1_3_2_1_20_1","volume-title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms. arXiv:1911.10635 [cs, stat] (April","author":"Zhang Kaiqing","year":"2021","unstructured":"Kaiqing Zhang , Zhuoran Yang , and Tamer Ba\u015far . 2021. Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms. arXiv:1911.10635 [cs, stat] (April 2021 ). http:\/\/arxiv.org\/abs\/1911.10635 arXiv:1911.10635. Kaiqing Zhang, Zhuoran Yang, and Tamer Ba\u015far. 2021. Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms. arXiv:1911.10635 [cs, stat] (April 2021). http:\/\/arxiv.org\/abs\/1911.10635 arXiv:1911.10635."}],"event":{"name":"PCI 2021: 25th Pan-Hellenic Conference on Informatics","acronym":"PCI 2021","location":"Volos Greece"},"container-title":["25th Pan-Hellenic Conference on Informatics"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503823.3503905","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503823.3503905","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:31Z","timestamp":1750188631000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503823.3503905"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,11,26]]},"references-count":20,"alternative-id":["10.1145\/3503823.3503905","10.1145\/3503823"],"URL":"https:\/\/doi.org\/10.1145\/3503823.3503905","relation":{},"subject":[],"published":{"date-parts":[[2021,11,26]]},"assertion":[{"value":"2022-02-22","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}