{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T15:15:09Z","timestamp":1784906109136,"version":"3.55.0"},"reference-count":21,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2017,10,1]],"date-time":"2017-10-01T00:00:00Z","timestamp":1506816000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"DOI":"10.13039\/501100001711","name":"Swiss National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001711","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100011021","name":"National Centre of Competence in Research Robotics","doi-asserted-by":"crossref","award":["200021-166232"],"award-info":[{"award-number":["200021-166232"]}],"id":[{"id":"10.13039\/501100011021","id-type":"DOI","asserted-by":"crossref"}]},{"name":"European Unions Horizon 2020 research and innovation programme","award":["644227"],"award-info":[{"award-number":["644227"]}]},{"name":"Swiss State Secretariat for Education, Research and Innovation (SERI)","award":["15.0029"],"award-info":[{"award-number":["15.0029"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2017,10]]},"DOI":"10.1109\/lra.2017.2720851","type":"journal-article","created":{"date-parts":[[2017,6,28]],"date-time":"2017-06-28T18:08:45Z","timestamp":1498673325000},"page":"2096-2103","source":"Crossref","is-referenced-by-count":502,"title":["Control of a Quadrotor With Reinforcement Learning"],"prefix":"10.1109","volume":"2","author":[{"given":"Jemin","family":"Hwangbo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Inkyu","family":"Sa","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Roland","family":"Siegwart","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marco","family":"Hutter","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref11","article-title":"High-dimensional continuous control using generalized advantage estimation","author":"schulman","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref12","volume":"1","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177703732"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729694"},{"key":"ref15","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","author":"duan","year":"0","journal-title":"Proc 33rd Int Conf Mach Learn"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2011.08.003"},{"key":"ref17","article-title":"Linear vs nonlinear MPC for trajectory tracking applied to rotary wing micro aerial vehicles","author":"kamel","year":"2016"},{"key":"ref18","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-319-54927-9_1","article-title":"Model predictive control for trajectory tracking of unmanned aerial vehicles using robot operating system","volume":"2","author":"kamel","year":"2017","journal-title":"Robot Operating System (ROS) The Complete Reference"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2013.6696917"},{"key":"ref4","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2005.1545025"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487175"},{"key":"ref8","first-page":"387","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"0","journal-title":"Proc 31st Int Conf Mach Learn"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1162\/089976698300017746"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2015.7353768"},{"key":"ref9","first-page":"1754","article-title":"Approximate dynamic programming finally performs well in the game of tetris","author":"gabillon","year":"2013","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980343"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2012.6224896"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7083369\/7951151\/07961277.pdf?arnumber=7961277","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T17:00:25Z","timestamp":1642006825000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7961277\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10]]},"references-count":21,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/lra.2017.2720851","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,10]]}}}