{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T06:07:38Z","timestamp":1783490858732,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,3,11]],"date-time":"2024-03-11T00:00:00Z","timestamp":1710115200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,3,11]]},"DOI":"10.1145\/3610977.3634984","type":"proceedings-article","created":{"date-parts":[[2024,3,10]],"date-time":"2024-03-10T00:19:00Z","timestamp":1710029940000},"page":"725-733","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Autonomous Assessment of Demonstration Sufficiency via Bayesian Inverse Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2373-1469","authenticated-orcid":false,"given":"Tu","family":"Trinh","sequence":"first","affiliation":[{"name":"University of California, Berkeley, Berkeley, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8493-9144","authenticated-orcid":false,"given":"Haoyu","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Utah, Salt Lake City, UT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9570-1832","authenticated-orcid":false,"given":"Daniel S.","family":"Brown","sequence":"additional","affiliation":[{"name":"University of Utah, Salt Lake City, UT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,3,11]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_2_2_1","unstructured":"Kareem Amin and Satinder Singh. 2016. Towards resolving unidentifiability in inverse reinforcement learning. arXiv preprint arXiv:1601.06569."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Brenna D Argall Sonia Chernova Manuela Veloso and Brett Browning. 2009. A survey of robot learning from demonstration. Robotics and autonomous systems 57 5 469--483.","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103500"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cognition.2009.07.005"},{"key":"e_1_3_2_2_6_1","unstructured":"Andr\u00e9 Barreto Will Dabney R\u00e9mi Munos Jonathan J. Hunt Tom Schaul Hado van Hasselt and David Silver. 2018. Successor features for transfer in reinforcement learning. arXiv preprint arXiv:1606.05312v2."},{"key":"e_1_3_2_2_7_1","volume-title":"Learning reward functions from diverse sources of human feedback: optimally integrating demonstrations and preferences. CoRR, abs\/2006.14091. https : \/ \/ arxiv . org \/ abs \/ 2006 . 14091 arXiv","author":"Biyik Erdem","year":"2006","unstructured":"Erdem Biyik, Dylan P. Losey, Malayandi Palan, Nicholas C. Landolfi, Gleb Shevchuk, and Dorsa Sadigh. 2020. Learning reward functions from diverse sources of human feedback: optimally integrating demonstrations and preferences. CoRR, abs\/2006.14091. https : \/ \/ arxiv . org \/ abs \/ 2006 . 14091 arXiv: 2006.14091."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2014.7040156"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568162.3576989"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1177\/02783649221078031"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Ralph Allan Bradley and Milton E Terry. 1952. Rank analysis of incomplete block designs: i. the method of paired comparisons. Biometrika 39 3\/4 324-- 345.","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"e_1_3_2_2_12_1","unstructured":"Greg Brockman Vicki Cheung Ludwig Pettersson Jonas Schneider John Schulman Jie Tang and Wojciech Zaremba. 2016. Openai gym. (2016). arXiv: 1606.01540 [cs.LG]."},{"key":"e_1_3_2_2_13_1","volume-title":"International Conference on Machine Learning. PMLR, 1165--1177","author":"Brown Daniel","year":"2020","unstructured":"Daniel Brown, Russell Coleman, Ravi Srinivasan, and Scott Niekum. 2020. Safe imitation learning via fast bayesian reward inference from preferences. In International Conference on Machine Learning. PMLR, 1165--1177."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11755"},{"key":"e_1_3_2_2_15_1","volume-title":"Conference on Robot Learning. PMLR, 362--372","author":"Brown Daniel S","year":"2018","unstructured":"Daniel S Brown, Yuchen Cui, and Scott Niekum. 2018. Risk-aware active inverse reinforcement learning. In Conference on Robot Learning. PMLR, 362--372."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33017749"},{"key":"e_1_3_2_2_17_1","unstructured":"Neerincx Burghouts Huizing. 2022. Robotic self-assessment of competence."},{"key":"e_1_3_2_2_18_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence number 1.","volume":"26","author":"Cakmak Maya","year":"2012","unstructured":"Maya Cakmak and Manuel Lopes. 2012. Algorithmic and human teaching of sequential decision tasks. In Proceedings of the AAAI Conference on Artificial Intelligence number 1. Vol. 26, 1536--1542."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/HRI.2013.6483603"},{"key":"e_1_3_2_2_20_1","volume-title":"International conference on machine learning. PMLR, 49--58","author":"Finn Chelsea","year":"2016","unstructured":"Chelsea Finn, Sergey Levine, and Pieter Abbeel. 2016. Guided cost learning: deep inverse optimal control via policy optimization. In International conference on machine learning. PMLR, 49--58."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Gaurav R Ghosal Matthew Zurek Daniel S Brown and Anca D Dragan. 2022. The effect of modeling human rationality level on learning rewards from multiple feedback types. arXiv preprint arXiv:2208.10687.","DOI":"10.1609\/aaai.v37i5.25740"},{"key":"e_1_3_2_2_22_1","volume-title":"Proceedings of the 31st annual conference of the cognitive science society. Cognitive Science Society Amsterdam, 2759--2764","author":"Goodman Noah D","year":"2009","unstructured":"Noah D Goodman, Chris L Baker, and Joshua B Tenenbaum. 2009. Cause and intent: social reasoning in causal learning. In Proceedings of the 31st annual conference of the cognitive science society. Cognitive Science Society Amsterdam, 2759--2764."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Noah D Goodman and Andreas Stuhlm\u00fcller. 2013. Knowledge and implicature: modeling language understanding as social cognition. Topics in cognitive science 5 1 173--184.","DOI":"10.1111\/tops.12007"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2016.7745138"},{"key":"e_1_3_2_2_25_1","first-page":"4415","article-title":"Reward-rational (implicit) choice: a unifying formalism for reward learning","volume":"33","author":"Jeon Hong Jun","year":"2020","unstructured":"Hong Jun Jeon, Smitha Milli, and Anca Dragan. 2020. Reward-rational (implicit) choice: a unifying formalism for reward learning. Advances in Neural Information Processing Systems, 33, 4415--4426.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9562048"},{"key":"e_1_3_2_2_27_1","unstructured":"Philippe Jorion. 2000. Value at risk."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Parameswaran Kamalaruban Rati Devidze Volkan Cevher and Adish Singla. 2019. Interactive teaching algorithms for inverse reinforcement learning. arXiv preprint arXiv:1905.11867.","DOI":"10.24963\/ijcai.2019\/374"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2010.06.005"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Pallavi Koppol Henny Admoni and Reid Simmons. 2021. Interaction considerations in learning from humans. In IJCAI.","DOI":"10.24963\/ijcai.2021\/40"},{"key":"e_1_3_2_2_31_1","volume-title":"International Conference on Learning Representations.","author":"Laidlaw Cassidy","year":"2021","unstructured":"Cassidy Laidlaw and Anca Dragan. 2021. The boltzmann policy distribution: accounting for systematic suboptimality in human models. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_32_1","unstructured":"Sergey Levine Zoran Popovic and Vladlen Koltun. 2011. Nonlinear inverse reinforcement learning with gaussian processes. Advances in neural information processing systems 24."},{"key":"e_1_3_2_2_33_1","volume-title":"Individual choice behavior: A theoretical analysis","author":"Luce R Duncan","unstructured":"R Duncan Luce. 2012. Individual choice behavior: A theoretical analysis. Courier Corporation."},{"key":"e_1_3_2_2_34_1","volume-title":"Conference on Robot Learning. PMLR, 1678--1690","author":"Ajay","unstructured":"Ajay Mandlekar et al. 2022. What matters in learning from offline human demonstrations for robot manipulation. In Conference on Robot Learning. PMLR, 1678--1690."},{"key":"e_1_3_2_2_35_1","first-page":"2","article-title":"Algorithms for inverse reinforcement learning","volume":"1","author":"Ng Andrew Y","year":"2000","unstructured":"Andrew Y Ng, Stuart Russell, et al. 2000. Algorithms for inverse reinforcement learning. In ICML. Vol. 1, 2.","journal-title":"ICML."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3522579"},{"key":"e_1_3_2_2_37_1","volume-title":"Hitting the target: stopping active learning at the cost-based optimum","author":"Pullar-Strecker Zac","unstructured":"Zac Pullar-Strecker, Katharina Dost, Eibe Frank, and J\u00f6rg Wicker. 2022. Hitting the target: stopping active learning at the cost-based optimum. In Springer."},{"key":"e_1_3_2_2_38_1","first-page":"2586","article-title":"Bayesian inverse reinforcement learning","volume":"7","author":"Ramachandran Deepak","year":"2007","unstructured":"Deepak Ramachandran and Eyal Amir. 2007. Bayesian inverse reinforcement learning. In IJCAI. Vol. 7, 2586--2591.","journal-title":"IJCAI."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Harish Ravichandar Athanasios S Polydoros Sonia Chernova and Aude Billard. 2020. Recent advances in robot learning from demonstration. Annual review of control robotics and autonomous systems 3 297--330.","DOI":"10.1146\/annurev-control-100819-063206"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Mariah L Schrum Erin Hedlund-Botti Nina Moorman and Matthew C Gombolay. 2022. Mind meld: personalized meta-learning for robot-centric imitation learning. In HRI 157--165.","DOI":"10.1109\/HRI53351.2022.9889616"},{"key":"e_1_3_2_2_41_1","article-title":"Benchmarks and algorithms for offline preference-based reward learning","author":"Shin Daniel","year":"2022","unstructured":"Daniel Shin, Anca Dragan, and Daniel S Brown. 2022. Benchmarks and algorithms for offline preference-based reward learning. Transactions on Machine Learning Research.","journal-title":"Transactions on Machine Learning Research."},{"key":"e_1_3_2_2_42_1","unstructured":"Umar Syed and Robert E Schapire. 2007. A game-theoretic approach to apprenticeship learning. Advances in Neural Information Processing Systems 20."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.5555\/2888116.2888133"},{"key":"e_1_3_2_2_44_1","unstructured":"Markus Wulfmeier Peter Ondruska and Ingmar Posner. 2015. Maximum entropy deep inverse reinforcement learning. arXiv preprint arXiv:1507.04888."},{"key":"e_1_3_2_2_45_1","first-page":"10496","article-title":"Curriculum design for teaching via demonstrations: theory and applications","volume":"34","author":"Yengera Gaurav","year":"2021","unstructured":"Gaurav Yengera, Rati Devidze, Parameswaran Kamalaruban, and Adish Singla. 2021. Curriculum design for teaching via demonstrations: theory and applications. Advances in Neural Information Processing Systems, 34, 10496--10509.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_46_1","volume-title":"Conference on Robot Learning. PMLR, 537--546","author":"Zakka Kevin","year":"2022","unstructured":"Kevin Zakka, Andy Zeng, Pete Florence, Jonathan Tompson, Jeannette Bohg, and Debidatta Dwibedi. 2022. Xirl: cross-embodiment inverse reinforcement learning. In Conference on Robot Learning. PMLR, 537--546."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Shao Zhifei and Er Meng Joo. 2012. A survey of inverse reinforcement learning techniques. International Journal of Intelligent Computing and Cybernetics.","DOI":"10.1109\/CEC.2012.6256507"},{"key":"e_1_3_2_2_48_1","unstructured":"Brian D Ziebart Andrew L Maas J Andrew Bagnell Anind K Dey et al. 2008. Maximum entropy inverse reinforcement learning. In Aaai. Vol. 8. Chicago IL USA 1433--1438."}],"event":{"name":"HRI '24: ACM\/IEEE International Conference on Human-Robot Interaction","location":"Boulder CO USA","acronym":"HRI '24","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3634984","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610977.3634984","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T16:35:41Z","timestamp":1756398941000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3634984"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,11]]},"references-count":48,"alternative-id":["10.1145\/3610977.3634984","10.1145\/3610977"],"URL":"https:\/\/doi.org\/10.1145\/3610977.3634984","relation":{},"subject":[],"published":{"date-parts":[[2024,3,11]]},"assertion":[{"value":"2024-03-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}