{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T15:54:22Z","timestamp":1783007662109,"version":"3.54.5"},"reference-count":25,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/icra55743.2025.11127461","type":"proceedings-article","created":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T17:28:56Z","timestamp":1756834136000},"page":"16869-16875","source":"Crossref","is-referenced-by-count":3,"title":["Safe Multi-Agent Navigation Guided by Goal-Conditioned Safe Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Meng","family":"Feng","sequence":"first","affiliation":[{"name":"Massachusetts Institute of Technology,Computer Science and Artificial Intelligence Laboratory,Cambridge,01239"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Viraj","family":"Parimi","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology,Computer Science and Artificial Intelligence Laboratory,Cambridge,01239"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Brian","family":"Williams","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology,Computer Science and Artificial Intelligence Laboratory,Cambridge,01239"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/socs.v10i1.18510"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v27i1.8541"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10801571"},{"key":"ref4","author":"Mirowski","year":"2017","journal-title":"Learning to navigate in complex environments"},{"key":"ref5","first-page":"1312","article-title":"Universal value function approximators","volume-title":"Proceedings of the 32nd International Conference on Machine Learning, ser. Proceedings of Machine Learning Research","volume":"37","author":"Schaul"},{"key":"ref6","volume-title":"Temporal difference models: Model-free deep rl for model-based control","author":"Pong","year":"2020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_179075"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_179075"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2899918"},{"key":"ref10","author":"Lynch","year":"2019","journal-title":"Learning latent plans from play"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8463162"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"ref13","article-title":"Search on the replay buffer: Bridging planning and reinforcement learning","volume":"32","author":"Eysenbach","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref14","volume-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2019"},{"key":"ref15","author":"Ray","year":"2019","journal-title":"Benchmarking Safe Exploration in Deep Reinforcement Learning"},{"key":"ref16","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"International conference on machine learning.","author":"Schulman","year":"2015"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.12794\/metadc1505267"},{"key":"ref18","author":"Sutton","year":"2018","journal-title":"Reinforcement learning: An introduction."},{"key":"ref19","first-page":"449","article-title":"A distributional per-spective on reinforcement learning","volume-title":"International conference on machine learning.","author":"Bellemare","year":"2017"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2014.11.006"},{"key":"ref21","volume-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","author":"Lowe","year":"2020"},{"key":"ref22","volume-title":"The surprising effectiveness of ppo in cooperative, multi-agent games","author":"Yu","year":"2022"},{"key":"ref23","volume-title":"The Replica dataset: A digital replica of indoor spaces","author":"Straub","year":"2019"},{"key":"ref24","article-title":"Habitat 2.0: Training home assistants to rearrange their habitat","author":"Szot","year":"2021","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"}],"event":{"name":"2025 IEEE International Conference on Robotics and Automation (ICRA)","location":"Atlanta, GA, USA","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,23]]}},"container-title":["2025 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11127273\/11127223\/11127461.pdf?arnumber=11127461","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T06:50:55Z","timestamp":1756882255000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11127461\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":25,"URL":"https:\/\/doi.org\/10.1109\/icra55743.2025.11127461","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}