{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T05:16:15Z","timestamp":1775193375862,"version":"3.50.1"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,12,14]],"date-time":"2020-12-14T00:00:00Z","timestamp":1607904000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,12,14]],"date-time":"2020-12-14T00:00:00Z","timestamp":1607904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,12,14]],"date-time":"2020-12-14T00:00:00Z","timestamp":1607904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,12,14]]},"DOI":"10.1109\/cdc42340.2020.9304030","type":"proceedings-article","created":{"date-parts":[[2021,1,13]],"date-time":"2021-01-13T07:27:32Z","timestamp":1610522852000},"page":"4099-4104","source":"Crossref","is-referenced-by-count":11,"title":["Coverage Path Planning with Proximal Policy Optimization in a Grid-based Environment"],"prefix":"10.1109","author":[{"given":"Aleksandr","family":"Ianenko","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexander","family":"Artamonov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Georgii","family":"Sarapulov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexey","family":"Safaraleev","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sergey","family":"Bogomolov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong-ki","family":"Noh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1177\/0278364907087426"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/21.148426"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1155\/2018\/5781591"},{"key":"ref13","author":"zhu","year":"2016","journal-title":"Target-driven visual navigation in indoor scenes using deep reinforcement learning"},{"key":"ref14","article-title":"Inverse Reinforcement Learning with Locally Consistent Reward Functions","author":"nguyen","year":"2015","journal-title":"Technical Report"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1177\/0278364917722396"},{"key":"ref16","article-title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","author":"schulman","year":"2015"},{"key":"ref17","article-title":"Proximal Policy Optimization Algorithms","author":"schulman","year":"2017"},{"key":"ref18","author":"schulman","year":"2015","journal-title":"Trust region policy optimization"},{"key":"ref19","author":"salimans","year":"2016","journal-title":"Weight normalization A simple reparameterization to accelerate training of deep neural networks"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.1985.1087316"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/s004220050262"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2013.09.004"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2008.2000394"},{"key":"ref7","first-page":"187","article-title":"Collision-free trajectory planning using distance transforms","volume":"10","author":"jarvis","year":"1985","journal-title":"Mechanical Engineering Trans of the Institution of Engineers Australia"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-4022-9_5"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2002.1013479"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/s10846-018-0787-7"},{"key":"ref20","author":"bhatt","year":"2019","journal-title":"CrossNorm Normalization for Off-Policy TD Reinforcement Learning"}],"event":{"name":"2020 59th IEEE Conference on Decision and Control (CDC)","location":"Jeju, Korea (South)","start":{"date-parts":[[2020,12,14]]},"end":{"date-parts":[[2020,12,18]]}},"container-title":["2020 59th IEEE Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9303728\/9303729\/09304030.pdf?arnumber=9304030","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T15:54:43Z","timestamp":1656345283000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9304030\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,12,14]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/cdc42340.2020.9304030","relation":{},"subject":[],"published":{"date-parts":[[2020,12,14]]}}}