{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T13:52:27Z","timestamp":1777038747012,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,3,9]],"date-time":"2020-03-09T00:00:00Z","timestamp":1583712000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100007187","name":"Lincoln Laboratory, Massachusetts Institute of Technology","doi-asserted-by":"publisher","award":["7000437192"],"award-info":[{"award-number":["7000437192"]}],"id":[{"id":"10.13039\/100007187","id-type":"DOI","asserted-by":"publisher"}]},{"name":"GaTech institute funding"},{"DOI":"10.13039\/100007297","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-18-S-B001"],"award-info":[{"award-number":["N00014-18-S-B001"]}],"id":[{"id":"10.13039\/100007297","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,3,9]]},"DOI":"10.1145\/3319502.3374791","type":"proceedings-article","created":{"date-parts":[[2020,3,7]],"date-time":"2020-03-07T01:30:31Z","timestamp":1583544631000},"page":"659-668","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Joint Goal and Strategy Inference across Heterogeneous Demonstrators via Reward Network Distillation"],"prefix":"10.1145","author":[{"given":"Letian","family":"Chen","sequence":"first","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rohan","family":"Paleja","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muyleng","family":"Ghuy","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Matthew","family":"Gombolay","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,3,9]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_2_2_1","first-page":"I","article-title":"Repeated Inverse Reinforcement Learning","volume":"30","author":"Amin Kareem","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_3_1","unstructured":"Dario Amodei Chris Olah Jacob Steinhardt Paul Christiano John Schulman and Dan Man\u00e9. 2016. Concrete Problems in AI Safety. arXiv preprint arXiv:1606.06565 (2016). https:\/\/arxiv.org\/pdf\/1606.06565.pdf  Dario Amodei Chris Olah Jacob Steinhardt Paul Christiano John Schulman and Dan Man\u00e9. 2016. Concrete Problems in AI Safety. arXiv preprint arXiv:1606.06565 (2016). https:\/\/arxiv.org\/pdf\/1606.06565.pdf"},{"key":"e_1_3_2_2_4_1","unstructured":"Greg Brockman Vicki Cheung Ludwig Pettersson Jonas Schneider John Schulman Jie Tang and Wojciech Zaremba. 2016. OpenAI Gym. arXiv:arXiv:1606.01540  Greg Brockman Vicki Cheung Ludwig Pettersson Jonas Schneider John Schulman Jie Tang and Wojciech Zaremba. 2016. OpenAI Gym. arXiv:arXiv:1606.01540"},{"key":"e_1_3_2_2_5_1","unstructured":"Daniel S. Brown Wonjoon Goo and Scott Niekum. 2019. Ranking-Based Reward Extrapolation without Rankings. CoRR abs\/1907.03976 (2019). arXiv:1907.03976 http:\/\/arxiv.org\/abs\/1907.03976  Daniel S. Brown Wonjoon Goo and Scott Niekum. 2019. Ranking-Based Reward Extrapolation without Rankings. CoRR abs\/1907.03976 (2019). arXiv:1907.03976 http:\/\/arxiv.org\/abs\/1907.03976"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CBMS.2006.87"},{"key":"e_1_3_2_2_7_1","volume-title":"Advances in Neural Information Processing Systems 25","author":"Choi Jaedeug"},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of Machine Learning Research (Proceedings of Machine Learning Research), Kamalika Chaudhuri and Masashi Sugiyama (Eds.)","volume":"89","author":"Czarnecki Wojciech M.","year":"2019"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1098\/rstb.2013.0478"},{"key":"e_1_3_2_2_10_1","unstructured":"Christos Dimitrakakis and Constantin A. Rothkopf. 2012. Bayesian Multitask Inverse Reinforcement Learning (EWRL'11). Springer-Verlag Berlin Heidelberg 273--284. https:\/\/doi.org\/10.1007\/978--3--642--29946--9_27  Christos Dimitrakakis and Constantin A. Rothkopf. 2012. Bayesian Multitask Inverse Reinforcement Learning (EWRL'11). Springer-Verlag Berlin Heidelberg 273--284. https:\/\/doi.org\/10.1007\/978--3--642--29946--9_27"},{"key":"e_1_3_2_2_11_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id= SJx63jRqFm","author":"Eysenbach Benjamin","year":"2019"},{"key":"e_1_3_2_2_12_1","volume-title":"International Conference on Machine Learning. 49--58","author":"Finn Chelsea","year":"2016"},{"key":"e_1_3_2_2_13_1","volume-title":"Learning Robust Rewards with Adverserial Inverse Reinforcement Learning. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkHywl-A-","author":"Fu Justin","year":"2018"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.11233"},{"key":"e_1_3_2_2_15_1","volume-title":"Proceedings of the Twenty-Fifth International Joint Conference on Artificial Intelligence. AAAI Press, 826--833","author":"Gombolay Matthew","year":"2016"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364918778344"},{"key":"e_1_3_2_2_17_1","volume-title":"Proceedings of the 35th International Conference on Machine Learning (Proceedings of Machine Learning Research), Jennifer Dy and Andreas Krause (Eds.)","volume":"80","author":"Haarnoja Tuomas","year":"2018"},{"key":"e_1_3_2_2_18_1","first-page":"I","article-title":"Multi-Modal Imitation Learning from Unstructured Demonstrations using Generative Adversarial Nets","volume":"30","author":"Hausman Karol","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_19_1","volume-title":"Thirty-Second AAAI Conference on Artificial Intelligence.","author":"Henderson Peter","year":"2018"},{"key":"e_1_3_2_2_20_1","unstructured":"Geoffrey Hinton Oriol Vinyals and Jeffrey Dean. 2015. Distilling the Knowledge in a Neural Network. In NIPS Deep Learning and Representation LearningWorkshop. http:\/\/arxiv.org\/abs\/1503.02531  Geoffrey Hinton Oriol Vinyals and Jeffrey Dean. 2015. Distilling the Knowledge in a Neural Network. In NIPS Deep Learning and Representation LearningWorkshop. http:\/\/arxiv.org\/abs\/1503.02531"},{"key":"e_1_3_2_2_21_1","volume-title":"Advances in Neural Information Processing Systems 29","author":"Ho Jonathan"},{"key":"e_1_3_2_2_22_1","unstructured":"Gary Klein. 1997. The recognition-primed decision (RPD) model: Looking back looking forward. Naturalistic decision making (1997) 285--292.  Gary Klein. 1997. The recognition-primed decision (RPD) model: Looking back looking forward. Naturalistic decision making (1997) 285--292."},{"key":"e_1_3_2_2_23_1","unstructured":"Gary A Klein Judith Ed Orasanu Roberta Ed Calderwood and Caroline E Zsambok. 1993. Decision making in action: Models and methods.. In This book is an outcome of a workshop held in Dayton OH Sep 25--27 1989. Ablex Publishing.  Gary A Klein Judith Ed Orasanu Roberta Ed Calderwood and Caroline E Zsambok. 1993. Decision making in action: Models and methods.. In This book is an outcome of a workshop held in Dayton OH Sep 25--27 1989. Ablex Publishing."},{"key":"e_1_3_2_2_24_1","first-page":"1","article-title":"End-to-end Training of Deep Visuomotor Policies","volume":"17","author":"Levine Sergey","year":"2016","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_2_25_1","first-page":"I","article-title":"InfoGAIL: Interpretable Imitation Learning from Visual Demonstrations","volume":"30","author":"Li Yunzhu","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_26_1","first-page":"2","article-title":"Algorithms for inverse reinforcement learning","volume":"1","author":"Ng Andrew Y","year":"2000","journal-title":"Icml"},{"key":"e_1_3_2_2_27_1","unstructured":"Stefanos Nikolaidis Keren Gu Ramya Ramakrishnan and Julie A. Shah. 2014. Efficient Model Learning for Human-Robot Collaborative Tasks. CoRR abs\/1405.6341 (2014). arXiv:1405.6341 http:\/\/arxiv.org\/abs\/1405.6341  Stefanos Nikolaidis Keren Gu Ramya Ramakrishnan and Julie A. Shah. 2014. Efficient Model Learning for Human-Robot Collaborative Tasks. CoRR abs\/1405.6341 (2014). arXiv:1405.6341 http:\/\/arxiv.org\/abs\/1405.6341"},{"key":"e_1_3_2_2_28_1","volume-title":"Game-Theoretic Modeling of Human Adaptation in Human-Robot Collaboration (HRI '17)","author":"Nikolaidis Stefanos","year":"2017"},{"key":"e_1_3_2_2_29_1","first-page":"1655","article-title":"Active learning with feedback on features and instances","author":"Raghavan Hema","year":"2006","journal-title":"Journal of Machine Learning Research 7"},{"key":"e_1_3_2_2_30_1","first-page":"2586","article-title":"Bayesian Inverse Reinforcement Learning","volume":"7","author":"Ramachandran Deepak","year":"2007","journal-title":"IJCAI"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143936"},{"key":"e_1_3_2_2_32_1","volume-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics (Proceedings of Machine Learning Research), Geoffrey Gordon, David Dunson, and Miroslav Dud\u00edk (Eds.)","volume":"15","author":"Ross Stephane","year":"2011"},{"key":"e_1_3_2_2_33_1","unstructured":"Andrei A Rusu Sergio Gomez Colmenarejo Caglar Gulcehre Guillaume Desjardins James Kirkpatrick Razvan Pascanu Volodymyr Mnih Koray Kavukcuoglu and Raia Hadsell. 2015. Policy distillation. arXiv preprint arXiv:1511.06295 (2015).  Andrei A Rusu Sergio Gomez Colmenarejo Caglar Gulcehre Guillaume Desjardins James Kirkpatrick Razvan Pascanu Volodymyr Mnih Koray Kavukcuoglu and Raia Hadsell. 2015. Policy distillation. arXiv preprint arXiv:1511.06295 (2015)."},{"key":"e_1_3_2_2_34_1","volume-title":"The adaptive web","author":"Schafer J Ben"},{"key":"e_1_3_2_2_35_1","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. CoRR abs\/1707.06347 (2017). arXiv:1707.06347 http:\/\/arxiv.org\/abs\/1707.06347  John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. CoRR abs\/1707.06347 (2017). arXiv:1707.06347 http:\/\/arxiv.org\/abs\/1707.06347"},{"key":"e_1_3_2_2_36_1","unstructured":"Herbert A Simon. 1972. Theories of bounded rationality. Decision and organization 1 1 (1972) 161--176.  Herbert A Simon. 1972. Theories of bounded rationality. Decision and organization 1 1 (1972) 161--176."},{"key":"e_1_3_2_2_37_1","first-page":"I","article-title":"Distral: Robust multitask reinforcement learning","volume":"30","author":"Teh Yee","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"e_1_3_2_2_39_1","unstructured":"Oriol Vinyals Igor Babuschkin Wojciech M Czarnecki Micha\u00ebl Mathieu Andrew Dudzik Junyoung Chung David H Choi Richard Powell Timo Ewalds Petko Georgiev etal 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature (2019) 1--5.  Oriol Vinyals Igor Babuschkin Wojciech M Czarnecki Micha\u00ebl Mathieu Andrew Dudzik Junyoung Chung David H Choi Richard Powell Timo Ewalds Petko Georgiev et al. 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature (2019) 1--5."},{"key":"e_1_3_2_2_40_1","unstructured":"Bryce Woodworth Francesco Ferrari Teofilo E. Zosa and Laurel D. Riek. 2018. Preference Learning in Assistive Robotics: Observational Repeated Inverse Reinforcement Learning. In Proceedings of the 3rd Machine Learning for Healthcare Conference (Proceedings of Machine Learning Research) Finale Doshi-Velez Jim Fackler Ken Jung David Kale Rajesh Ranganath Byron Wallace and Jenna Wiens (Eds.) Vol. 85. PMLR Palo Alto California 420--439. http:\/\/proceedings.mlr.press\/v85\/woodworth18a.html  Bryce Woodworth Francesco Ferrari Teofilo E. Zosa and Laurel D. Riek. 2018. Preference Learning in Assistive Robotics: Observational Repeated Inverse Reinforcement Learning. In Proceedings of the 3rd Machine Learning for Healthcare Conference (Proceedings of Machine Learning Research) Finale Doshi-Velez Jim Fackler Ken Jung David Kale Rajesh Ranganath Byron Wallace and Jenna Wiens (Eds.) Vol. 85. PMLR Palo Alto California 420--439. http:\/\/proceedings.mlr.press\/v85\/woodworth18a.html"},{"key":"e_1_3_2_2_41_1","unstructured":"MarkusWulfmeier Peter Ondruska and Ingmar Posner. 2015. Maximum entropy deep inverse reinforcement learning. arXiv preprint arXiv:1507.04888 (2015).  MarkusWulfmeier Peter Ondruska and Ingmar Posner. 2015. Maximum entropy deep inverse reinforcement learning. arXiv preprint arXiv:1507.04888 (2015)."},{"key":"e_1_3_2_2_42_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning (Proceedings of Machine Learning Research), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.)","volume":"97","author":"Xu Kelvin","year":"2019"},{"key":"e_1_3_2_2_43_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning (Proceedings of Machine Learning Research), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.)","volume":"97","author":"Zhang Yunbo","year":"2019"},{"key":"e_1_3_2_2_44_1","unstructured":"Brian D. Ziebart. 2010. Modeling Purposeful Adaptive Behavior with the Principle of Maximum Causal Entropy. Ph.D. Dissertation. Pittsburgh PA USA. Advisor(s) Bagnell J. Andrew. AAI3438449.  Brian D. Ziebart. 2010. Modeling Purposeful Adaptive Behavior with the Principle of Maximum Causal Entropy. Ph.D. Dissertation. Pittsburgh PA USA. Advisor(s) Bagnell J. Andrew. AAI3438449."},{"key":"e_1_3_2_2_45_1","unstructured":"Brian D. Ziebart Andrew Maas J. Andrew Bagnell and Anind K. Dey. 2008. Maximum Entropy Inverse Reinforcement Learning (AAAI'08). AAAI Press 1433-- 1438. http:\/\/dl.acm.org\/citation.cfm?id=1620270.1620297  Brian D. Ziebart Andrew Maas J. Andrew Bagnell and Anind K. Dey. 2008. Maximum Entropy Inverse Reinforcement Learning (AAAI'08). AAAI Press 1433-- 1438. http:\/\/dl.acm.org\/citation.cfm?id=1620270.1620297"}],"event":{"name":"HRI '20: ACM\/IEEE International Conference on Human-Robot Interaction","location":"Cambridge United Kingdom","acronym":"HRI '20","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction","IEEE-RAS Robotics and Automation"]},"container-title":["Proceedings of the 2020 ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3319502.3374791","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3319502.3374791","content-type":"text\/html","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3319502.3374791","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3319502.3374791","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:38:21Z","timestamp":1750199901000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3319502.3374791"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,3,9]]},"references-count":45,"alternative-id":["10.1145\/3319502.3374791","10.1145\/3319502"],"URL":"https:\/\/doi.org\/10.1145\/3319502.3374791","relation":{},"subject":[],"published":{"date-parts":[[2020,3,9]]},"assertion":[{"value":"2020-03-09","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}