{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T21:11:25Z","timestamp":1780607485065,"version":"3.54.1"},"reference-count":49,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,5,30]],"date-time":"2021-05-30T00:00:00Z","timestamp":1622332800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,5,30]],"date-time":"2021-05-30T00:00:00Z","timestamp":1622332800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,5,30]]},"DOI":"10.1109\/icra48506.2021.9561891","type":"proceedings-article","created":{"date-parts":[[2021,10,20]],"date-time":"2021-10-20T00:28:35Z","timestamp":1634689715000},"page":"6214-6221","source":"Crossref","is-referenced-by-count":21,"title":["Learning Dense Rewards for Contact-Rich Manipulation Tasks"],"prefix":"10.1109","author":[{"given":"Zheng","family":"Wu","sequence":"first","affiliation":[{"name":"X, the Moonshot Factory,Mountain View,CA,USA,94043"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenzhao","family":"Lian","sequence":"additional","affiliation":[{"name":"X, the Moonshot Factory,Mountain View,CA,USA,94043"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vaibhav","family":"Unhelkar","sequence":"additional","affiliation":[{"name":"Rice University,Houston,TX,USA,77005"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masayoshi","family":"Tomizuka","sequence":"additional","affiliation":[{"name":"University of California, Berkeley,Berkeley,CA,USA,94704"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stefan","family":"Schaal","sequence":"additional","affiliation":[{"name":"X, the Moonshot Factory,Mountain View,CA,USA,94043"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793485"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.aav3123"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8462891"},{"key":"ref32","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"levine","year":"2016","journal-title":"The Journal of Machine Learning Research"},{"key":"ref31","first-page":"arxiv-1909","article-title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning","author":"cabi","year":"2019"},{"key":"ref30","article-title":"Random expert distillation: Imitation learning via expert policy support estimation","author":"wang","year":"2019"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989326"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794219"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2011.6095096"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.009"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/s00422-014-0599-1"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.3005126"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2013.6630743"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref1","author":"sutton","year":"2018","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"ref22","first-page":"2","article-title":"Algorithms for inverse reinforcement learning","volume":"1","author":"ng","year":"2000","journal-title":"ICML"},{"key":"ref21","first-page":"12","article-title":"Robot learning from demonstration","volume":"97","author":"atkeson","year":"1997","journal-title":"ICML"},{"key":"ref24","first-page":"4565","article-title":"Generative adversarial imitation learning","author":"ho","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref23","article-title":"Maximum margin planning","author":"bagnell","year":"2006","journal-title":"Proceedings of the International Conference on Machine Learning (ICML)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2015.7139555"},{"key":"ref25","first-page":"8538","article-title":"Variational inverse control with events: A general framework for data-driven reward definition","author":"fu","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref10","article-title":"Learning robust rewards with adversarial inverse reinforcement learning","author":"fu","year":"2017"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2019.XV.073"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2015.XI.044"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196582"},{"key":"ref13","first-page":"2234","article-title":"Improved techniques for training gans","author":"salimans","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref14","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"haarnoja","year":"2018","journal-title":"International Conference on Machine Learning"},{"key":"ref15","article-title":"User-agent value alignment","author":"shapiro","year":"2002","journal-title":"Proc of The 18th Nat Conf on Artif Intell AAAI"},{"key":"ref16","article-title":"Concrete problems in ai safety","author":"amodei","year":"2016"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.2200\/S00568ED1V01Y201402AIM028"},{"key":"ref18","first-page":"6765","article-title":"Inverse reward design","author":"hadfield-menell","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref19","article-title":"Maximum likelihood constraint inference for inverse reinforcement learning","author":"scobee","year":"2019","journal-title":"International Conference on Learning Representations"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"883","DOI":"10.1109\/LRA.2020.2965869","article-title":"Benchmarking protocols for evaluating small parts robotic assembly systems","volume":"5","author":"kimble","year":"2020","journal-title":"IEEE l of Robotics and Automation"},{"key":"ref3","first-page":"651","article-title":"Scalable deep reinforcement learning for vision-based robotic manipulation","author":"kalashnikov","year":"2018","journal-title":"Conference on Robot Learning"},{"key":"ref6","first-page":"2586","article-title":"Bayesian inverse reinforcement learning","volume":"7","author":"ramachandran","year":"2007","journal-title":"IJCAI"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref8","first-page":"2672","article-title":"Generative adversarial nets","author":"goodfellow","year":"2014","journal-title":"Advances in neural information processing systems"},{"key":"ref7","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"8","author":"ziebart","year":"2008","journal-title":"AAAI"},{"key":"ref49","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014"},{"key":"ref9","first-page":"49","article-title":"Guided cost learning: Deep inverse optimal control via policy optimization","author":"finn","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref46","article-title":"Active learning literature survey","author":"settles","year":"2009","journal-title":"University of Wisconsin-Madison Computer Sciences Department Tech Report"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196733"},{"key":"ref48","article-title":"Gpu-accelerated robotic simulation for distributed reinforcement learning","author":"liang","year":"2018"},{"key":"ref47","article-title":"Wavenet: A generative model for raw audio","author":"oord","year":"2016"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8968201"},{"key":"ref41","author":"puterman","year":"2014","journal-title":"Markov Decision Processes Discrete Stochastic Dynamic Programming"},{"key":"ref44","article-title":"Reverse curriculum generation for reinforcement learning","author":"florensa","year":"2017"},{"key":"ref43","first-page":"2601","article-title":"Where do rewards come from","author":"singh","year":"2009","journal-title":"Proceedings of the Annual Conference of the Cognitive Science Society"}],"event":{"name":"2021 IEEE International Conference on Robotics and Automation (ICRA)","location":"Xi'an, China","start":{"date-parts":[[2021,5,30]]},"end":{"date-parts":[[2021,6,5]]}},"container-title":["2021 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9560720\/9560666\/09561891.pdf?arnumber=9561891","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,2]],"date-time":"2022-08-02T23:22:33Z","timestamp":1659482553000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9561891\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,30]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/icra48506.2021.9561891","relation":{},"subject":[],"published":{"date-parts":[[2021,5,30]]}}}