{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,31]],"date-time":"2025-05-31T09:24:16Z","timestamp":1748683456721,"version":"3.28.0"},"reference-count":30,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,5,25]]},"DOI":"10.23919\/acc50511.2021.9482906","type":"proceedings-article","created":{"date-parts":[[2021,7,28]],"date-time":"2021-07-28T20:29:16Z","timestamp":1627504156000},"page":"1959-1964","source":"Crossref","is-referenced-by-count":2,"title":["Online Observer-Based Inverse Reinforcement Learning"],"prefix":"10.23919","author":[{"given":"Ryan","family":"Self","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kevin","family":"Coleman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"He","family":"Bai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rushikesh","family":"Kamalapurkar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Online observer-based inverse reinforcement learning","year":"2020","author":"self","key":"ref30"},{"key":"ref10","first-page":"1342","article-title":"Feature construction for inverse reinforcement learning","author":"levine","year":"0","journal-title":"Advances in Neural Information Processing Systems 23"},{"key":"ref11","first-page":"19","article-title":"Nonlinear inverse reinforcement learning with Gaussian processes","author":"levine","year":"0","journal-title":"Advances in Neural Information Processing Systems 24"},{"key":"ref12","first-page":"1413","article-title":"Inverse reinforcement learning in swarm systems","author":"\u0161o\u0161i?","year":"0","journal-title":"Proc Conf Auton Agents MultiAgent Syst International Foundation for Autonomous Agents and Multiagent Systems"},{"key":"ref13","first-page":"5143","article-title":"Competitive multi-agent inverse reinforcement learning with sub-optimal demonstrations","volume":"80","author":"wang","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33486-3_10"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2010.02.018"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2015.2466191"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-78384-0"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619314"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2018.8431430"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2017.7963838"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1115\/1.3653115"},{"journal-title":"Matrix Analysis","year":"1993","author":"horn","key":"ref27"},{"key":"ref3","first-page":"554","article-title":"Inverse reinforcement learning","author":"abbeel","year":"2010","journal-title":"Encyclopedia of Machine Learning"},{"key":"ref6","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","author":"ziebart","year":"0","journal-title":"Proc AAAI Conf Artif Intel"},{"journal-title":"Online simultaneous state and parameter estimation","year":"0","author":"self","key":"ref29"},{"key":"ref5","first-page":"1040","article-title":"Learning from demonstration","author":"schaal","year":"1997","journal-title":"Advances in Neural Information Processing Systems 9"},{"journal-title":"Maximum entropy deep inverse reinforcement learning","year":"2015","author":"wulfmeier","key":"ref8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143936"},{"key":"ref2","first-page":"663","article-title":"Algorithms for inverse reinforcement learning","author":"ng","year":"2000","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016722"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/279943.279964"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CCTA.2019.8920458"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.23919\/ACC45564.2020.9147344"},{"key":"ref21","article-title":"Online inverse reinforcement learning with limited data","author":"self","year":"0","journal-title":"Proc IEEE Conf Decis Control"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1515\/9781400842643"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9029636"},{"journal-title":"Nonlinear Systems","year":"2002","author":"khalil","key":"ref26"},{"journal-title":"Adaptive Control Stability Convergence and Robustness","year":"1989","author":"sastry","key":"ref25"}],"event":{"name":"2021 American Control Conference (ACC)","start":{"date-parts":[[2021,5,25]]},"location":"New Orleans, LA, USA","end":{"date-parts":[[2021,5,28]]}},"container-title":["2021 American Control Conference (ACC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9482409\/9482614\/09482906.pdf?arnumber=9482906","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,11,1]],"date-time":"2021-11-01T20:54:41Z","timestamp":1635800081000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9482906\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,25]]},"references-count":30,"URL":"https:\/\/doi.org\/10.23919\/acc50511.2021.9482906","relation":{},"subject":[],"published":{"date-parts":[[2021,5,25]]}}}