{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T12:26:42Z","timestamp":1730204802145,"version":"3.28.0"},"reference-count":19,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/cdc40024.2019.9029286","type":"proceedings-article","created":{"date-parts":[[2020,3,13]],"date-time":"2020-03-13T04:43:11Z","timestamp":1584074591000},"page":"4609-4614","source":"Crossref","is-referenced-by-count":1,"title":["Transforming Policy via Reward Advancement"],"prefix":"10.1109","author":[{"given":"Guojun","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanhua","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume":"99","author":"ng","year":"1999","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref11","first-page":"1333","article-title":"Transfer in reinforcement learning via shared features","volume":"13","author":"konidaris","year":"2012","journal-title":"Journal of Machine Learning Research"},{"key":"ref12","first-page":"49","article-title":"Guided cost learning: Deep inverse optimal control via policy optimization","author":"finn","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611975673.88"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2018.00070"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2019.00194"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1190"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref18","first-page":"19","article-title":"Nonlinear inverse reinforcement learning with gaussian processes","author":"levine","year":"2011","journal-title":"Advances in neural information processing systems"},{"article-title":"Equivalence between policy gradients and soft q-learning","year":"2017","author":"schulman","key":"ref19"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2015.04.029"},{"journal-title":"Modeling purposeful adaptive behavior with the principle of maximum causal entropy (CMU PhD dissertation)","year":"2010","author":"ziebart","key":"ref3"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2018.00070"},{"key":"ref5","article-title":"Agent-based behavioral model of spatial learning and route choice","author":"zhang","year":"2006","journal-title":"Transportation Research Board 85th Annual Meeting"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/2629592"},{"key":"ref7","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"8","author":"ziebart","year":"2008","journal-title":"AAAI"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.jtrangeo.2014.08.006"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2017.8264410"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1123\/jpah.8.s1.s72"}],"event":{"name":"2019 IEEE 58th Conference on Decision and Control (CDC)","start":{"date-parts":[[2019,12,11]]},"location":"Nice, France","end":{"date-parts":[[2019,12,13]]}},"container-title":["2019 IEEE 58th Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8977134\/9028853\/09029286.pdf?arnumber=9029286","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,19]],"date-time":"2022-07-19T20:17:47Z","timestamp":1658261867000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9029286\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":19,"URL":"https:\/\/doi.org\/10.1109\/cdc40024.2019.9029286","relation":{},"subject":[],"published":{"date-parts":[[2019,12]]}}}