{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T16:26:48Z","timestamp":1779294408803,"version":"3.51.4"},"reference-count":28,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,8]]},"DOI":"10.1109\/roman.2018.8525837","type":"proceedings-article","created":{"date-parts":[[2018,11,8]],"date-time":"2018-11-08T18:29:37Z","timestamp":1541701777000},"page":"1156-1162","source":"Crossref","is-referenced-by-count":15,"title":["Interactive Reinforcement Learning from Demonstration and Human Evaluative Feedback"],"prefix":"10.1109","author":[{"given":"Guangliang","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Randy","family":"Gomez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keisuke","family":"Nakamura","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","author":"knox","year":"2012","journal-title":"Learning from Human-generated Reward"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1597735.1597738"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"5","DOI":"10.65109\/MCUE9477","article-title":"Combining manual feedback with subsequent mdp reward signals for reinforcement learning","author":"knox","year":"2010","journal-title":"Proc of International Conference on Autonomous Agents and Multiagent Systems"},{"key":"ref13","doi-asserted-by":"crossref","first-page":"475","DOI":"10.65109\/PAJO3896","article-title":"Reinforcement learning from simultaneous human and mdp reward","author":"knox","year":"2012","journal-title":"Proc of International Conference on Autonomous Agents and Multiagent Systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2015.03.009"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-015-9308-2"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-017-9374-8"},{"key":"ref17","doi-asserted-by":"crossref","DOI":"10.1609\/aaai.v28i1.8839","article-title":"A strategy-aware technique for learning behaviors from discrete human feedback","author":"loftin","year":"2014","journal-title":"Proceedings of the 28th AAAI Conference on Artificial Intelligence (AAAI-2014"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2014.6926319"},{"key":"ref19","author":"macglashan","year":"2017","journal-title":"Interactive learning from policy-dependent human feedback"},{"key":"ref28","first-page":"3","volume":"8","author":"watkins","year":"1992","journal-title":"Q-learning Machine Learning"},{"key":"ref4","article-title":"Reinforcement learning from demonstration through shaping","author":"brys","year":"2015","journal-title":"Proceedings of the International Joint Conference on Artificial Intelligence (IJCAI)"},{"key":"ref27","author":"warnell","year":"2017","journal-title":"Deep tamer Interactive agent shaping in high-dimensional state spaces"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"ref6","author":"howard","year":"1960","journal-title":"Dynamic Programming and Markov Processes"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1613\/jair.2584"},{"key":"ref8","first-page":"1890","article-title":"Imitation learning with demonstrations and shaping rewards","author":"judah","year":"2014","journal-title":"Proceedings of the 28th AAAI Conference on Artificial Intelligence"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/375735.376334"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/1228716.1228725"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-22362-4_31"},{"key":"ref20","first-page":"663","article-title":"Algorithms for inverse reinforcement learning","author":"ng","year":"2000","journal-title":"Proceedings of International Conference on Machine Learning (ICML)"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICORR.2011.5975338"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2011.6005273"},{"key":"ref24","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2011.6005223"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2007.09.009"},{"key":"ref25","first-page":"483","article-title":"Dynamic reward shaping: training a robot by voice","author":"tenorio-gonzalez","year":"2010","journal-title":"Advances in Artifical Intelligence IBERAMIA"}],"event":{"name":"2018 27th IEEE International Symposium on Robot and Human Interactive Communication (RO-MAN)","location":"Nanjing","start":{"date-parts":[[2018,8,27]]},"end":{"date-parts":[[2018,8,31]]}},"container-title":["2018 27th IEEE International Symposium on Robot and Human Interactive Communication (RO-MAN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8509495\/8525500\/08525837.pdf?arnumber=8525837","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T00:34:49Z","timestamp":1775262889000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8525837\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,8]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/roman.2018.8525837","relation":{},"subject":[],"published":{"date-parts":[[2018,8]]}}}