{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:54:40Z","timestamp":1773536080584,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,16]]},"DOI":"10.1145\/3757279.3785589","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:27:38Z","timestamp":1773102458000},"page":"187-195","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["What You Reward Is What You Learn: Comparing Rewards for Online Speech Policy Optimization in Public HRI"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8408-4437","authenticated-orcid":false,"given":"Sichao","family":"Song","sequence":"first","affiliation":[{"name":"CyberAgent, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9547-3681","authenticated-orcid":false,"given":"Yuki","family":"Okafuji","sequence":"additional","affiliation":[{"name":"CyberAgent, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6286-9906","authenticated-orcid":false,"given":"Kaito","family":"Ariu","sequence":"additional","affiliation":[{"name":"CyberAgent, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9088-6578","authenticated-orcid":false,"given":"Amy","family":"Koike","sequence":"additional","affiliation":[{"name":"University of Wisconsin-Madison, Madison, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,16]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Marcin Andrychowicz Filip Wolski Alex Ray Jonas Schneider Rachel Fong Peter Welinder Bob McGrew Josh Tobin Pieter Abbeel and Wojciech Zaremba. 2018. Hindsight Experience Replay. arxiv:1707.01495. arxiv:1707.01495"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11127795"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-25554-5_7"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-021-05983-y"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/HRI.2019.8673234"},{"key":"e_1_3_2_2_6_1","volume-title":"An empirical evaluation of thompson sampling. Advances in neural information processing systems, 24","author":"Chapelle Olivier","year":"2011","unstructured":"Olivier Chapelle and Lihong Li. 2011. An empirical evaluation of thompson sampling. Advances in neural information processing systems, 24 (2011)."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/tro.2020.2964824"},{"key":"e_1_3_2_2_8_1","volume-title":"Conference on Uncertainty in Artificial Intelligence. 181\u2013190","author":"Chen Yifang","year":"2020","unstructured":"Yifang Chen, Alex Cuellar, Haipeng Luo, Jignesh Modi, Heramb Nemlekar, and Stefanos Nikolaidis. 2020. Fair contextual multi-armed bandits: Theory and experiments. In Conference on Uncertainty in Artificial Intelligence. 181\u2013190."},{"key":"e_1_3_2_2_9_1","unstructured":"Paul Christiano Jan Leike Tom B. Brown Miljan Martic Shane Legg and Dario Amodei. 2023. Deep reinforcement learning from human preferences. arxiv:1706.03741. arxiv:1706.03741"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2019.00048"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2018.8525832"},{"key":"e_1_3_2_2_12_1","unstructured":"Koji Inoue Yuki Okafuji Jun Baba Yoshiki Ohira Katsuya Hyodo and Tatsuya Kawahara. 2025. A Noise-Robust Turn-Taking System for Real-World Dialogue Robots: A Field Experiment. arxiv:2503.06241. arxiv:2503.06241"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-34106-9_18"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.336"},{"key":"e_1_3_2_2_15_1","volume-title":"Time-Varying Preference Bandits for Robot Behavior Personalization. APPLIED SCIENCES-BASEL, 14, 23","author":"Kim Chanwoo","year":"2024","unstructured":"Chanwoo Kim, Joonhyeok Lee, Eunwoo Kim, and Kyungjae Lee. 2024. Time-Varying Preference Bandits for Robot Behavior Personalization. APPLIED SCIENCES-BASEL, 14, 23 (2024)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-03194-1_2"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/HRI61500.2025.10974089"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1017\/9781108571401"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","unstructured":"Guiying Liu Tatjana Korbanka Jasmin Timm Naomi Veit Laura Maier Markus Huff and Frank Papenmeier. 2025. The Bystander Effect in Human-Robot Interaction: How the Presence of Bystanders Inhibits Help for a Social Robot. Sep. https:\/\/doi.org\/10.23668\/psycharchives.21242 10.23668\/psycharchives.21242","DOI":"10.23668\/psycharchives.21242"},{"key":"e_1_3_2_2_21_1","volume-title":"Jos\u00e9 Carlos Castillo, \u00c1lvaro Castro-Gonz\u00e1lez, and Miguel \u00c1ngel Salichs","author":"Maroto-G\u00f3mez Marcos","year":"2024","unstructured":"Marcos Maroto-G\u00f3mez, Mar\u00eda Malfaz, Jos\u00e9 Carlos Castillo, \u00c1lvaro Castro-Gonz\u00e1lez, and Miguel \u00c1ngel Salichs. 2024. Personalizing activity selection in assistive social robots from explicit and implicit user feedback. International Journal of Social Robotics, 1\u201319."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2019.05.004"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","unstructured":"Zi Haur Pang Yahui Fu Divesh Lala Mikey Elmers Koji Inoue and Tatsuya Kawahara. 2025. Does the Appearance of Autonomous Conversational Robots Affect User Spoken Behaviors in Real-World Conference Interactions? In Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems (CHI EA \u201925). Association for Computing Machinery Article 198 8 pages. https:\/\/doi.org\/10.1145\/3706599.3720179 10.1145\/3706599.3720179","DOI":"10.1145\/3706599.3720179"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/HUMANOIDS.2016.7803357"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3279954.3279957"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3316782.3316791"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3316782.3316791"},{"key":"e_1_3_2_2_28_1","volume-title":"Abbas Kazerouni, Ian Osband, Zheng Wen, et al.","author":"Russo Daniel J","year":"2018","unstructured":"Daniel J Russo, Benjamin Van Roy, Abbas Kazerouni, Ian Osband, Zheng Wen, et al. 2018. A tutorial on thompson sampling. Foundations and Trends\u00ae in Machine Learning, 11, 1 (2018), 1\u201396."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2017.8172476"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2020.532279"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/25.3-4.285"},{"key":"e_1_3_2_2_32_1","volume-title":"Task Engagement as Personalization Feedback for Socially-Assistive Robots and Cognitive Training. Technologies, 6, 2","author":"Tsiakas Konstantinos","year":"2018","unstructured":"Konstantinos Tsiakas, Maher Abujelala, and Fillia Makedon. 2018. Task Engagement as Personalization Feedback for Socially-Assistive Robots and Cognitive Training. Technologies, 6, 2 (2018), issn:2227-7080 https:\/\/www.mdpi.com\/2227-7080\/6\/2\/49"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/1734454.1734468"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989121"}],"event":{"name":"HRI '26: 21st ACM\/IEEE International Conference on Human-Robot Interaction","location":"Edinburgh Scotland UK","acronym":"HRI '26","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction","IEEE RAS"]},"container-title":["Proceedings of the 21st ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:31:36Z","timestamp":1773534696000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757279.3785589"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,16]]},"references-count":34,"alternative-id":["10.1145\/3757279.3785589","10.1145\/3757279"],"URL":"https:\/\/doi.org\/10.1145\/3757279.3785589","relation":{},"subject":[],"published":{"date-parts":[[2026,3,16]]},"assertion":[{"value":"2026-03-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}