{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T11:06:03Z","timestamp":1762254363447,"version":"3.37.3"},"reference-count":37,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,23]],"date-time":"2022-10-23T00:00:00Z","timestamp":1666483200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/1000000010","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2132887"],"award-info":[{"award-number":["2132887"]}],"id":[{"id":"10.13039\/1000000010","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,23]]},"DOI":"10.1109\/iros47612.2022.9982282","type":"proceedings-article","created":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T19:38:15Z","timestamp":1672083495000},"page":"863-870","source":"Crossref","is-referenced-by-count":7,"title":["Keeping Humans in the Loop: Teaching via Feedback in Continuous Action Space Environments"],"prefix":"10.1109","author":[{"given":"Isaac","family":"Sheidlower","sequence":"first","affiliation":[{"name":"Tufts University School of Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Allison","family":"Moore","sequence":"additional","affiliation":[{"name":"Tufts University School of Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Elaine","family":"Short","sequence":"additional","affiliation":[{"name":"Tufts University School of Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"DQN-TAMER: Human-in-the-Loop Reinforcement Learning with Intractable Feedback","author":"Arakawa","year":"2018","journal-title":"arXiv"},{"key":"ref2","article-title":"Deep Reinforcement Learning from Policy-Dependent Human Feedback","author":"Arumugam","year":"2019","journal-title":"arXiv"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11485"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2019.2946162"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/1597735.1597738"},{"key":"ref7","first-page":"2625","article-title":"Policy Shaping: Integrating Human Feedback with Reinforcement Learning","volume-title":"Advances in Neural Information Processing Systems 26","author":"Griffith","year":"2013"},{"key":"ref8","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","volume-title":"Proc. of the 16th Int. Conf. on Machine Learning","author":"Ng"},{"key":"ref9","first-page":"10","author":"Wu","year":"2017","journal-title":"TRAINING AGENT FOR FIRST-PERSON SHOOTER GAME WITH ACTOR-CRITIC CURRICULUM LEARNING"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.3015448"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICTC49870.2020.9289596"},{"key":"ref12","article-title":"Reward Shaping via Meta-Learning","author":"Zou","year":"2019","journal-title":"arXiv"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10269"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23780-5_11"},{"article-title":"Deep reinforcement learning from human preferences","volume-title":"arXiv","author":"Christiano","key":"ref15"},{"key":"ref16","article-title":"Interactive Learning from Policy-Dependent Human Feedback","author":"MacGlashan","year":"2017","journal-title":"arXiv"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3406499.3418769"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-017-17682-7"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-012-0412-6"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/s10846-018-0839-z"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref22","article-title":"Hindsight Experience Replay","author":"Andrychowicz","year":"2018","journal-title":"arXiv"},{"key":"ref23","article-title":"Proximal Policy Optimization Algorithms","author":"Schulman","year":"2017","journal-title":"arXiv"},{"key":"ref24","article-title":"Addressing Function Approximation Error in Actor-Critic Methods","author":"Fujimoto","year":"2018","journal-title":"arXiv"},{"key":"ref25","article-title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","author":"Haarnoja","year":"2018","journal-title":"arXiv"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3319502.3374832"},{"journal-title":"Policy Shaping in Domains with Multiple Optimal Policies (Extended Abstract)","first-page":"2","author":"Sahni","key":"ref27"},{"key":"ref28","first-page":"429","article-title":"LESS is More: Rethinking Probabilistic Models of Human Behavior","volume-title":"Proc. of the 2020 ACM\/IEEE Int. Conf. on Human-Robot Interaction","author":"Bobu"},{"issue":"3","key":"ref29","doi-asserted-by":"crossref","first-page":"279","DOI":"10.1007\/BF00992698","article-title":"Technical Note: Q-Learning","volume":"8","author":"Watkins","year":"1992","journal-title":"Machine Learning"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3061308"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CoG47356.2020.9231687"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/3357236.3395525"},{"key":"ref33","article-title":"Soft Actor-Critic Algorithms and Applications","author":"Haarnoja","year":"2019","journal-title":"arXiv"},{"volume-title":"Flask web development: developing web applications with python","year":"2018","author":"Grinberg","key":"ref34"},{"journal-title":"Brax-A Differentiable Physics Engine for Large Scale Rigid Body Simulation","year":"2021","author":"Freeman","key":"ref35"},{"key":"ref36","article-title":"Off-Policy Deep Reinforce-ment Learning without Exploration","author":"Fujimoto","year":"2019","journal-title":"arXiv"},{"key":"ref37","article-title":"Stabilizing Off-Policy Q-Learning via Bootstrapping Error Reduction","author":"Kumar","year":"2019","journal-title":"arXiv"}],"event":{"name":"2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2022,10,23]]},"location":"Kyoto, Japan","end":{"date-parts":[[2022,10,27]]}},"container-title":["2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9981026\/9981028\/09982282.pdf?arnumber=9982282","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T03:30:55Z","timestamp":1706758255000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9982282\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,23]]},"references-count":37,"URL":"https:\/\/doi.org\/10.1109\/iros47612.2022.9982282","relation":{},"subject":[],"published":{"date-parts":[[2022,10,23]]}}}