{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T12:50:02Z","timestamp":1784638202099,"version":"3.55.0"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/USG.html"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1109\/icra40945.2020.9197158","type":"proceedings-article","created":{"date-parts":[[2020,9,15]],"date-time":"2020-09-15T21:25:46Z","timestamp":1600205146000},"page":"1364-1371","source":"Crossref","is-referenced-by-count":36,"title":["Feedback Linearization for Uncertain Systems via Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Tyler","family":"Westenbroek","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Fridovich-Keil","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eric","family":"Mazumdar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shreyas","family":"Arora","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Valmik","family":"Prabhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"S. Shankar","family":"Sastry","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Claire J.","family":"Tomlin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2001.933002"},{"key":"ref38","article-title":"Baxter","author":"robotics","year":"2013"},{"key":"ref33","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref32","author":"sutton","year":"1998","journal-title":"Introduction to Reinforcement Learning"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/9.1316"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2019.2958840"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ISMA.2009.5164788"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1119\/1.16860"},{"key":"ref35","article-title":"Deterministic Policy Gradient Algorithms","author":"silver","year":"2014","journal-title":"Proceedings of the 31st International Conference on Machine Learning Proceedings of Machine Learning Research"},{"key":"ref34","article-title":"Proximal Policy Optimization Algorithms","author":"schulman","year":"0","journal-title":"CoRR"},{"key":"ref10","article-title":"Contributions to the theory of optimal control","volume":"5","author":"kalman","year":"1960","journal-title":"Bol Soc Mat Mexicana"},{"key":"ref40","article-title":"Baxter Humanoid Robot Kinematics&#x00A9; 2017 Dr. Bob Productions Robert L. Williams II, Ph. D., williar4@ ohio. edu Mechanical Engineering, Ohio University, April 2017","author":"williams","year":"2017"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1017\/9781139061759"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/9.898695"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2014.2299335"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980409"},{"key":"ref15","author":"sastry","year":"1989","journal-title":"Adaptive Control Stability Convergence and Robustness"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1177\/027836498700600202"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/9.40741"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/9.1308"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.1991.4791451"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/9.995038"},{"key":"ref4","author":"sutton","year":"2018","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/9.754811"},{"key":"ref3","author":"bertsekas","year":"1996","journal-title":"Neuro-Dynamic Programming"},{"key":"ref6","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2008.929402"},{"key":"ref5","article-title":"Policy gradient methods for reinforcement learning with function approximation","author":"sutton","year":"2000","journal-title":"Advances in neural information processing systems"},{"key":"ref8","article-title":"Proximal policy optimization al-gorithms","author":"schulman","year":"2017"},{"key":"ref7","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2015"},{"key":"ref2","author":"isidori","year":"2013","journal-title":"Nonlinear Control Systems"},{"key":"ref9","article-title":"Flat sys-tems, equivalence and trajectory generation","author":"martin","year":"2003"},{"key":"ref1","volume":"10","author":"sastry","year":"1999","journal-title":"Nonlinear Systems Analysis Stability and Control"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2017.8264435"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2013.6580235"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2014.2319052"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/91.531775"},{"key":"ref41","article-title":"ROS: an Open-Source Robot Operating System","author":"quigley","year":"2009","journal-title":"ICRA Workshop on Open Source Software"},{"key":"ref23","article-title":"Feedback Linearization for Unknown Systems via Reinforcement Learning","author":"westenbroek","year":"2019"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICNN.1994.374620"},{"key":"ref25","article-title":"Adaptive control of a class of nonlinear discrete-time systems using neural networks","volume":"40","author":"chen","year":"1995","journal-title":"IEEE Transactions on Automatic Control"}],"event":{"name":"2020 IEEE International Conference on Robotics and Automation (ICRA)","location":"Paris, France","start":{"date-parts":[[2020,5,31]]},"end":{"date-parts":[[2020,8,31]]}},"container-title":["2020 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9187508\/9196508\/09197158.pdf?arnumber=9197158","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,28]],"date-time":"2022-06-28T00:27:23Z","timestamp":1656376043000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9197158\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/icra40945.2020.9197158","relation":{},"subject":[],"published":{"date-parts":[[2020,5]]}}}