{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T09:22:16Z","timestamp":1778318536424,"version":"3.51.4"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,23]]},"DOI":"10.1109\/iros47612.2022.9981319","type":"proceedings-article","created":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T19:38:15Z","timestamp":1672083495000},"page":"8814-8820","source":"Crossref","is-referenced-by-count":36,"title":["Transferring Multi-Agent Reinforcement Learning Policies for Autonomous Driving using Sim-to-Real"],"prefix":"10.1109","author":[{"given":"Eduardo","family":"Candela","sequence":"first","affiliation":[{"name":"Imperial College London,Centre for Transport Studies,Department of Civil and Environmental Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Leandro","family":"Parada","sequence":"additional","affiliation":[{"name":"Imperial College London,Centre for Transport Studies,Department of Civil and Environmental Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luis","family":"Marques","sequence":"additional","affiliation":[{"name":"Imperial College London,Centre for Transport Studies,Department of Civil and Environmental Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tiberiu-Andrei","family":"Georgescu","sequence":"additional","affiliation":[{"name":"Imperial College London,Centre for Transport Studies,Department of Civil and Environmental Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yiannis","family":"Demiris","sequence":"additional","affiliation":[{"name":"Imperial College London,Personal Robotics Laboratory,Department of Electrical and Electronic Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Panagiotis","family":"Angeloudis","sequence":"additional","affiliation":[{"name":"Imperial College London,Centre for Transport Studies,Department of Civil and Environmental Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Humans just cant stop rear-ending self-driving cars-lets figure out why","author":"Stewart","year":"2018"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9207663"},{"key":"ref3","article-title":"Smarts: Scalable multi-agent reinforcement learning training school for autonomous driving","author":"Zhou","year":"2020","journal-title":"arXiv preprint"},{"key":"ref4","doi-asserted-by":"crossref","DOI":"10.1109\/IROS.2017.8202133","article-title":"Domain randomization for transferring deep neural networks from simulation to the real world","volume-title":"CoRR, vol. abs\/1703.06907","author":"Tobin","year":"2017"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/SSCI47803.2020.9308468"},{"key":"ref6","article-title":"Leave no trace: Learning to reset for safe and autonomous reinforcement learning","author":"Eysenbach","year":"2017","journal-title":"arXiv preprint"},{"key":"ref7","doi-asserted-by":"crossref","DOI":"10.1109\/IROS45743.2020.9341260","volume-title":"Sim2Real Transfer for Reinforcement Learning without Dynamics Randomization","author":"Kaspar","year":"2020"},{"key":"ref8","volume-title":"Sim-to-Real Reinforcement Learning for Deformable Object Manipulation","author":"Matas","year":"2018"},{"key":"ref9","volume-title":"Learning to Play Soccer by Reinforcement and Applying Sim-to-Real to Compete in the Real World","author":"Bassani","year":"2020"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3366622.3368146"},{"key":"ref11","first-page":"10571","article-title":"Robust multi-agent reinforcement learning with model uncertainty","volume":"33","author":"Zhang","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989179"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-49116-3_38"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21236\/ada637949"},{"key":"ref15","article-title":"R-maddpg for partially observable environments and limited communication","author":"Wang","year":"2020","journal-title":"arXiv preprint"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6226"},{"key":"ref17","first-page":"1","article-title":"Carla: An open urban driving simulator","volume-title":"Conference on robot learning","author":"Dosovitskiy"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1111\/mice.12702"},{"key":"ref19","article-title":"The ingredients of real-world robotic reinforcement learning","author":"Zhu","year":"2020","journal-title":"arXiv preprint"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561384"},{"key":"ref21","article-title":"Model: A modularized end-to-end reinforcement learning framework for autonomous driving","author":"Wang","year":"2021","journal-title":"arXiv preprint"},{"key":"ref22","article-title":"Deepracer: Educational autonomous racing platform for experimentation with sim2real reinforcement learning","author":"Balaji","year":"2019","journal-title":"arXiv preprint"},{"key":"ref23","article-title":"The surprising effectiveness of ppo in cooperative, multi-agent games","author":"Yu","year":"2021","journal-title":"arXiv preprint"},{"key":"ref24","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume":"30","author":"Lowe","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref25","first-page":"4295","article-title":"Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"International Conference on Machine Learning","author":"Rashid"},{"key":"ref26","article-title":"DuckieBot MOOC Founders edition datasheet","author":"Duckietown","year":"2020","journal-title":"DB21-M datasheet"},{"key":"ref27","volume-title":"Duckietown environments for openai gym","author":"Chevalier-Boisvert","year":"2018"},{"key":"ref28","article-title":"Modelling strategies for a mobile robot","volume-title":"Masters thesis, Swiss Federal Institute of Technology Zurich","author":"Ercan","year":"2019"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/0191-2615(86)90012-3"},{"key":"ref30","article-title":"On a for-mal model of safe and scalable self-driving cars","author":"Shalev-Shwartz","year":"2017","journal-title":"arXiv preprint"}],"event":{"name":"2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Kyoto, Japan","start":{"date-parts":[[2022,10,23]]},"end":{"date-parts":[[2022,10,27]]}},"container-title":["2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9981026\/9981028\/09981319.pdf?arnumber=9981319","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,26]],"date-time":"2024-06-26T17:54:38Z","timestamp":1719424478000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9981319\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,23]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/iros47612.2022.9981319","relation":{},"subject":[],"published":{"date-parts":[[2022,10,23]]}}}