{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T17:10:05Z","timestamp":1756401005673,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,3,11]],"date-time":"2024-03-11T00:00:00Z","timestamp":1710115200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"US National Science Foundation","award":["IIS 2132887"],"award-info":[{"award-number":["IIS 2132887"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,3,11]]},"DOI":"10.1145\/3610977.3634947","type":"proceedings-article","created":{"date-parts":[[2024,3,10]],"date-time":"2024-03-10T00:19:00Z","timestamp":1710029940000},"page":"639-648","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Online Behavior Modification for Expressive User Control of RL-Trained Robots"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4487-8357","authenticated-orcid":false,"given":"Isaac","family":"Sheidlower","sequence":"first","affiliation":[{"name":"Tufts University, Medford, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5289-7098","authenticated-orcid":false,"given":"Mavis","family":"Murdock","sequence":"additional","affiliation":[{"name":"Tufts University, Medford, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8653-4474","authenticated-orcid":false,"given":"Emma","family":"Bethel","sequence":"additional","affiliation":[{"name":"Tufts University, Medford, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8615-3715","authenticated-orcid":false,"given":"Reuben M.","family":"Aronson","sequence":"additional","affiliation":[{"name":"Tufts University, Medford, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4789-9979","authenticated-orcid":false,"given":"Elaine Schaertl","family":"Short","sequence":"additional","affiliation":[{"name":"Tufts University, Medford, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,3,11]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","unstructured":"Mohammed Alshiekh Roderick Bloem Ruediger Ehlers Bettina K\u00f6nighofer Scott Niekum and Ufuk Topcu. 2017. Safe Reinforcement Learning via Shielding. https:\/\/doi.org\/10.48550\/arXiv.1708.08611 arXiv:1708.08611 [cs].","DOI":"10.48550\/arXiv.1708.08611"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1201\/9781315140223"},{"key":"e_1_3_2_2_3_1","unstructured":"Marcin Andrychowicz Filip Wolski Alex Ray Jonas Schneider Rachel Fong Peter Welinder Bob McGrew Josh Tobin Pieter Abbeel and Wojciech Zaremba. 2018. Hindsight Experience Replay. http:\/\/arxiv.org\/abs\/1707.01495 arXiv:1707.01495 [cs]."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357236.3395525"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3171221.3171284"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","unstructured":"Erdem Biyik Nicolas Huynh Mykel Kochenderfer and Dorsa Sadigh. 2020. Active Preference-Based Gaussian Process Regression for Reward Learning. In Robotics: Science and Systems XVI. Robotics: Science and Systems Foundation. https:\/\/doi.org\/10.15607\/RSS.2020.XVI.041","DOI":"10.15607\/RSS.2020.XVI.041"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3434073.3444667"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568162.3576983"},{"key":"e_1_3_2_2_9_1","unstructured":"Greg Brockman Vicki Cheung Ludwig Pettersson Jonas Schneider John Schulman Jie Tang and Wojciech Zaremba. 2016. OpenAI Gym. http:\/\/arxiv.org\/abs\/1606.01540 arXiv:1606.01540 [cs]."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1177\/02783649211041652"},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning. PMLR, 1430--1440","author":"Chane-Sane Elliot","year":"2021","unstructured":"Elliot Chane-Sane, Cordelia Schmid, and Ivan Laptev. 2021. Goal-Conditioned Reinforcement Learning with Imagined Subgoals. In Proceedings of the 38th International Conference on Machine Learning. PMLR, 1430--1440. https:\/\/proceedings.mlr.press\/v139\/chane-sane21a.html ISSN: 2640--3498."},{"key":"e_1_3_2_2_12_1","unstructured":"Paul Christiano Jan Leike Tom B. Brown Miljan Martic Shane Legg and Dario Amodei. 2017. Deep reinforcement learning from human preferences. http:\/\/arxiv.org\/abs\/1706.03741 Number: arXiv:1706.03741 arXiv:1706.03741 [cs stat]."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6385907"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3321707.3321804"},{"key":"e_1_3_2_2_15_1","unstructured":"Benjamin Eysenbach Abhishek Gupta Julian Ibarz and Sergey Levine. 2018. Diversity is All You Need: Learning Skills without a Reward Function. http:\/\/arxiv.org\/abs\/1802.06070 Number: arXiv:1802.06070 arXiv:1802.06070 [cs]."},{"key":"e_1_3_2_2_16_1","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Eysenbach Benjamin","year":"2022","unstructured":"Benjamin Eysenbach, Tianjun Zhang, Sergey Levine, and Russ R. Salakhutdinov. 2022. Contrastive Learning as Goal-Conditioned Reinforcement Learning. Advances in Neural Information Processing Systems , Vol. 35 (Dec. 2022), 35603--35620. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/hash\/e7663e974c4ee7a2b475a4775201ce1f-Abstract-Conference.html"},{"key":"e_1_3_2_2_17_1","volume-title":"Fontaine and Stefanos Nikolaidis","author":"Matthew","year":"2021","unstructured":"Matthew C. Fontaine and Stefanos Nikolaidis. 2021. Differentiable Quality Diversity. http:\/\/arxiv.org\/abs\/2106.03894 Number: arXiv:2106.03894 arXiv:2106.03894 [cs]."},{"key":"e_1_3_2_2_18_1","unstructured":"Javier Garcia and Fernando Fernandez. [n. d.]. A Comprehensive Survey on Safe Reinforcement Learning. ( [n. d.])."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2016.2593928"},{"key":"e_1_3_2_2_20_1","volume-title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor. arXiv:1801.01290 [cs, stat] (Aug","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor. arXiv:1801.01290 [cs, stat] (Aug. 2018). http:\/\/arxiv.org\/abs\/1801.01290"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364918776060"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","unstructured":"Shervin Javdani Siddhartha Srinivasa and Andrew Bagnell. 2015. Shared Autonomy via Hindsight Optimization. In Robotics: Science and Systems XI. Robotics: Science and Systems Foundation. https:\/\/doi.org\/10.15607\/RSS.2015.XI.032","DOI":"10.15607\/RSS.2015.XI.032"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/THMS.2018.2878815"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2909824.3020249"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1597735.1597738"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2004.1389727"},{"key":"e_1_3_2_2_27_1","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Kumar Saurabh","year":"2020","unstructured":"Saurabh Kumar, Aviral Kumar, Sergey Levine, and Chelsea Finn. 2020. One Solution is Not All You Need: Few-Shot Extrapolation via Structured MaxEnt RL. In Advances in Neural Information Processing Systems, Vol. 33. Curran Associates, Inc., 8198--8210. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/5d151d1059a6281335a10732fc49620e-Abstract.html"},{"key":"e_1_3_2_2_28_1","unstructured":"Minghuan Liu Menghui Zhu and Weinan Zhang. 2022. Goal-Conditioned Reinforcement Learning: Problems and Solutions. http:\/\/arxiv.org\/abs\/2201.08299 arXiv:2201.08299 [cs]."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793611"},{"key":"e_1_3_2_2_30_1","volume-title":"Proceedings of The 2nd Conference on Robot Learning. PMLR, 879--893","author":"Mandlekar Ajay","year":"2018","unstructured":"Ajay Mandlekar, Yuke Zhu, Animesh Garg, Jonathan Booher, Max Spero, Albert Tung, Julian Gao, John Emmons, Anchit Gupta, Emre Orbay, Silvio Savarese, and Li Fei-Fei. 2018. ROBOTURK: A Crowdsourcing Platform for Robotic Skill Learning through Imitation. In Proceedings of The 2nd Conference on Robot Learning. PMLR, 879--893. https:\/\/proceedings.mlr.press\/v87\/mandlekar18a.html ISSN: 2640--3498."},{"key":"e_1_3_2_2_31_1","volume-title":"Margolis and Pulkit Agrawal","author":"Gabriel","year":"2022","unstructured":"Gabriel B. Margolis and Pulkit Agrawal. 2022. Walk These Ways: Tuning Robot Control for Generalization with Multiplicity of Behavior. http:\/\/arxiv.org\/abs\/2212.03238 arXiv:2212.03238 [cs, eess]."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3128237"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196582"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","unstructured":"Christopher Mower Joao Moura and Sethu Vijayakumar. 2021. Skill-based Shared Control. In Robotics: Science and Systems XVII. Robotics: Science and Systems Foundation. https:\/\/doi.org\/10.15607\/RSS.2021.XVII.028","DOI":"10.15607\/RSS.2021.XVII.028"},{"key":"e_1_3_2_2_35_1","volume-title":"Proceedings of The 2nd Conference on Robot Learning. PMLR, 700--713","author":"Muratore Fabio","year":"2018","unstructured":"Fabio Muratore, Felix Treede, Michael Gienger, and Jan Peters. 2018. Domain Randomization for Simulation-Based Policy Optimization with Transferability Assessment. In Proceedings of The 2nd Conference on Robot Learning. PMLR, 700--713. https:\/\/proceedings.mlr.press\/v87\/muratore18a.html ISSN: 2640--3498."},{"key":"e_1_3_2_2_36_1","volume-title":"Proceedings of the 5th Conference on Robot Learning. PMLR, 342--352","author":"Myers Vivek","year":"2022","unstructured":"Vivek Myers, Erdem Biyik, Nima Anari, and Dorsa Sadigh. 2022. Learning Multimodal Rewards from Rankings. In Proceedings of the 5th Conference on Robot Learning. PMLR, 342--352. https:\/\/proceedings.mlr.press\/v164\/myers22a.html ISSN: 2640--3498."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568162.3576965"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1177\/02783649211050677"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2022.04.009"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.mechatronics.2010.04.005"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980259"},{"key":"e_1_3_2_2_42_1","volume-title":"Stanley","author":"Pugh Justin K.","year":"2016","unstructured":"Justin K. Pugh, Lisa B. Soros, and Kenneth O. Stanley. 2016. Quality Diversity: A New Frontier for Evolutionary Computation. Frontiers in Robotics and AI , Vol. 3 (2016). https:\/\/www.frontiersin.org\/article\/10.3389\/frobt.2016.00040"},{"key":"e_1_3_2_2_43_1","volume-title":"Dragan","author":"Ratner Ellis","year":"2018","unstructured":"Ellis Ratner, Dylan Hadfield-Menell, and Anca D. Dragan. 2018. Simplifying Reward Design through Divide-and-Conquer. http:\/\/arxiv.org\/abs\/1806.02501 Number: arXiv:1806.02501 arXiv:1806.02501 [cs]."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"crossref","unstructured":"Siddharth Reddy Anca D. Dragan and Sergey Levine. 2018. Shared Autonomy via Deep Reinforcement Learning. http:\/\/arxiv.org\/abs\/1802.01744 Number: arXiv:1802.01744 arXiv:1802.01744 [cs].","DOI":"10.15607\/RSS.2018.XIV.005"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3100603"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9982282"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Bryon Tjanaka Matthew C. Fontaine Julian Togelius and Stefanos Nikolaidis. 2022. Approximating Gradients for Differentiable Quality Diversity in Reinforcement Learning. http:\/\/arxiv.org\/abs\/2202.03666 Number: arXiv:2202.03666 arXiv:2202.03666 [cs].","DOI":"10.1145\/3512290.3528705"},{"key":"e_1_3_2_2_48_1","unstructured":"Bryon Tjanaka Matthew C. Fontaine Yulun Zhang Sam Sommerer Nathan Dennler and Stefanos Nikolaidis. 2021. pyribs: A bare-bones Python library for quality diversity optimization. https:\/\/github.com\/icaros-usc\/pyribs Publication Title: GitHub repository."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","unstructured":"Josh Tobin Rachel Fong Alex Ray Jonas Schneider Wojciech Zaremba and Pieter Abbeel. 2017. Domain Randomization for Transferring Deep Neural Networks from Simulation to the Real World. https:\/\/doi.org\/10.48550\/arXiv.1703.06907 arXiv:1703.06907 [cs].","DOI":"10.48550\/arXiv.1703.06907"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.3758\/s13423-020-01798--5"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.5555\/3523760.3523825"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.2307\/30036540"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3319502.3374821"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2022.05.014"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2021.103199"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561839"}],"event":{"name":"HRI '24: ACM\/IEEE International Conference on Human-Robot Interaction","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction"],"location":"Boulder CO USA","acronym":"HRI '24"},"container-title":["Proceedings of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3634947","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3610977.3634947","content-type":"text\/html","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610977.3634947","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3610977.3634947","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T16:31:36Z","timestamp":1756398696000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3610977.3634947"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,11]]},"references-count":56,"alternative-id":["10.1145\/3610977.3634947","10.1145\/3610977"],"URL":"https:\/\/doi.org\/10.1145\/3610977.3634947","relation":{},"subject":[],"published":{"date-parts":[[2024,3,11]]},"assertion":[{"value":"2024-03-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}