{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:38:40Z","timestamp":1783150720793,"version":"3.54.6"},"reference-count":76,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,8,1]],"date-time":"2019-08-01T00:00:00Z","timestamp":1564617600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["51809246"],"award-info":[{"award-number":["51809246"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2018QF003"],"award-info":[{"award-number":["ZR2018QF003"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"crossref","award":["841713015"],"award-info":[{"award-number":["841713015"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["861705020036"],"award-info":[{"award-number":["861705020036"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Honda Research Institute Japan Co. Ltd."}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Human-Mach. Syst."],"published-print":{"date-parts":[[2019,8]]},"DOI":"10.1109\/thms.2019.2912447","type":"journal-article","created":{"date-parts":[[2019,5,8]],"date-time":"2019-05-08T00:51:30Z","timestamp":1557276690000},"page":"337-349","source":"Crossref","is-referenced-by-count":105,"title":["Human-Centered Reinforcement Learning: A Survey"],"prefix":"10.1109","volume":"49","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1728-5711","authenticated-orcid":false,"given":"Guangliang","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Randy","family":"Gomez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keisuke","family":"Nakamura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9092-9152","authenticated-orcid":false,"given":"Bo","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1038\/518486a"},{"key":"ref71","first-page":"1","article-title":"Face valuing: Training user interfaces with facial expressions and reinforcement learning","author":"veeriah","year":"0","journal-title":"Proc Mach Learning Int Workshop"},{"key":"ref70","first-page":"1353","article-title":"Towards learning from implicit human reward","author":"li","year":"0","journal-title":"Proc 10th Int Conf Auton Agents Multiagent Syst"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1016\/j.cognition.2017.03.006"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2014.09.003"},{"key":"ref39","first-page":"475","article-title":"Reinforcement learning from simultaneous human and MDP reward","author":"knox","year":"0","journal-title":"Int Conf Auton Agents Multiagent syst International Foundation for Autonomous Agents and Multiagent Systems"},{"key":"ref75","first-page":"4299","article-title":"Deep reinforcement learning from human preferences","author":"christiano","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-25085-9_65"},{"key":"ref33","first-page":"3366","article-title":"Policy shaping with human teachers","author":"cederborg","year":"0","journal-title":"Proc 24th Int Joint Conf Artif Intell"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-015-9283-7"},{"key":"ref31","first-page":"937","article-title":"A strategy-aware technique for learning behaviors from discrete human feedback","author":"loftin","year":"0","journal-title":"Proc 28th AAAI Conf Artif Intell"},{"key":"ref30","first-page":"1","article-title":"Reinforcement learning combined with human feedback in continuous state and action spaces","author":"vien","year":"0","journal-title":"Proc IEEE Int Conf Develop Learn Epigenetic Robot"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/S0921-8890(02)00168-9"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/566654.566597"},{"key":"ref35","first-page":"2285","article-title":"Interactive learning from policy-dependent human feedback","volume":"70","author":"macglashan","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref34","first-page":"2625","article-title":"Policy shaping: Integrating human feedback with reinforcement learning","author":"griffith","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/DEVLRN.2014.6982960"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-22362-4_31"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-017-9374-8"},{"key":"ref63","first-page":"113","article-title":"Challenges to decoding the intention behind natural instruction","author":"peralta","year":"0","journal-title":"Proc IEEE Int Symp Robot Human Interact Commun"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICORR.2011.5975338"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2018.8525837"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2015.03.009"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref66","first-page":"909","article-title":"Using informative behavior to increase engagement in the TAMER framework","author":"li","year":"0","journal-title":"Proc Int Conf Auton Agents and Multi Agent Syst"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2011.6005223"},{"key":"ref67","author":"vail","year":"1994","journal-title":"Emotion The On\/Off Switch for Learning"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-72348-6_6"},{"key":"ref69","first-page":"1771","article-title":"A large-scale study of agents learning from human reward","author":"li","year":"0","journal-title":"Proc 10th Int Conf Auton Agents Multiagent Syst"},{"key":"ref2","first-page":"539","article-title":"Learning and sequential decision making (technical report)","author":"barto","year":"1989","journal-title":"Learning and Computational Neuroscience"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1162\/106365602320169811"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-16952-6_49"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2007.09.009"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/375735.376334"},{"key":"ref23","first-page":"1000","article-title":"Reinforcement learning with human teachers: Evidence of feedback and guidance with implications for learning performance","author":"thomaz","year":"0","journal-title":"Proc 20th AAAI Conf Artif Intell"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/1597735.1597738"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-006-0005-z"},{"key":"ref50","article-title":"Combining probabilities","author":"bailer-jones","year":"2011"},{"key":"ref51","first-page":"761","article-title":"Bayesian Q-learning","author":"dearden","year":"0","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s12369-012-0163-x"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2008.4543752"},{"key":"ref57","first-page":"1","article-title":"Real-time interactive reinforcement learning for robots","author":"thomaz","year":"0","journal-title":"AAAI Workshop on Human Comprehensible Machine Learning"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2005.1545011"},{"key":"ref55","first-page":"1","article-title":"Transparency and socially guided machine learning","author":"thomaz","year":"0","journal-title":"Proc 5th Int Conf Develop Learn"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-015-9308-2"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2009.07.008"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.conb.2008.08.003"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCC.2012.2218595"},{"key":"ref40","article-title":"Learning from human-generated reward","author":"knox","year":"2012"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1038\/nature14540"},{"key":"ref13","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","author":"sutton","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref14","first-page":"406","article-title":"Pegasus: A policy search method for large MDPs and POMDPs","author":"ng","year":"0","journal-title":"Proc 16th Conf Uncertainty Artif Intell"},{"key":"ref15","first-page":"489","article-title":"Learning to cooperate via policy search","author":"peshkin","year":"0","journal-title":"Proc 16th Conf Uncertainty Artif Intell"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2004.1307456"},{"key":"ref17","first-page":"465","article-title":"Pilco: A model-based and data-efficient approach to policy search","author":"deisenroth","year":"0","journal-title":"Proc 28th Int Conf Mach Learn"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1613\/jair.613"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/5.784219"},{"key":"ref4","first-page":"278","article-title":"Policy invariance under reward transformations: Theory and application to reward shaping","author":"ng","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IS.2008.4670492"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273572"},{"key":"ref8","first-page":"920","article-title":"Teaching with rewards and punishments: Reinforcement or communication","author":"ho","year":"0","journal-title":"Proc 37th Ann Meeting Cog Sci Soc"},{"key":"ref7","first-page":"2652","article-title":"Expressing arbitrary reward functions as potential-based advice","author":"harutyunyan","year":"0","journal-title":"Proc 29th AAAI Conf Artif Intell"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1111\/j.2517-6161.1977.tb01600.x"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1613\/jair.301"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/2449396.2449422"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-02675-6_46"},{"key":"ref48","author":"skinner","year":"1953","journal-title":"Science and Human Behavior"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1037\/0278-7393.10.4.598"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/203330.203343"},{"key":"ref44","first-page":"5","article-title":"Combining manual feedback with subsequent MDP reward signals for reinforcement learning","author":"knox","year":"0","journal-title":"Proc 9th Int Conf Auton Agents Multiagent Syst"},{"key":"ref43","article-title":"Socially intelligent autonomous agents that learn from human reward","author":"li","year":"2016"}],"container-title":["IEEE Transactions on Human-Machine Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6221037\/8763787\/08708686.pdf?arnumber=8708686","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T21:07:15Z","timestamp":1657746435000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8708686\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,8]]},"references-count":76,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/thms.2019.2912447","relation":{},"ISSN":["2168-2291","2168-2305"],"issn-type":[{"value":"2168-2291","type":"print"},{"value":"2168-2305","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,8]]}}}