{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T11:19:20Z","timestamp":1773141560719,"version":"3.50.1"},"reference-count":32,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,8,25]],"date-time":"2025-08-25T00:00:00Z","timestamp":1756080000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,8,25]]},"DOI":"10.1109\/ro-man63969.2025.11217743","type":"proceedings-article","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T18:42:29Z","timestamp":1762195349000},"page":"636-642","source":"Crossref","is-referenced-by-count":1,"title":["Exploring Unstructured Language Feedback for Robot Learning"],"prefix":"10.1109","author":[{"given":"Hannah","family":"Kuehn","sequence":"first","affiliation":[{"name":"KTH Royal Institute of Technology,Division of Robotics, Perception, and Learning,Stockholm,Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"William","family":"Ahlberg","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology,Division of Robotics, Perception, and Learning,Stockholm,Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joseph La","family":"Delfa","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology,Division of Robotics, Perception, and Learning,Stockholm,Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Iolanda","family":"Leite","sequence":"additional","affiliation":[{"name":"KTH Royal Institute of Technology,Division of Robotics, Perception, and Learning,Stockholm,Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","author":"Brohan","year":"2023"},{"key":"ref2","doi-asserted-by":"crossref","first-page":"63","DOI":"10.7551\/mitpress\/7221.001.0001","volume-title":"Where the Action Is: The Foundations of Embodied Interaction","author":"Dourish","year":"2001"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2021.646002"},{"key":"ref4","first-page":"217","article-title":"Learning robot objectives from physical human interaction","volume-title":"Proceedings of the 1st Annual Conference on Robot Learning","author":"Bajcsy"},{"key":"ref5","first-page":"141","article-title":"Learning from physical human corrections, one feature at a time","volume-title":"Proceedings of the 2018 ACM\/IEEE International Conference on Human-Robot Interaction, HRI \u201918","author":"Bajcsy"},{"key":"ref6","first-page":"796","article-title":"Learning under misspecified objective spaces","volume-title":"Proceedings of The 2nd Conference on Robot Learning","author":"Bobu"},{"key":"ref7","article-title":"Deep reinforcement learning from human preferences","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Christiano"},{"key":"ref8","first-page":"6152","article-title":"PEBBLE: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Lee"},{"key":"ref9","first-page":"79","article-title":"Design principles for creating human-shapable agents","volume-title":"AAAI Spring Symposium: Agents that Learn from Human Teachers","author":"Knox"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11485"},{"key":"ref11","first-page":"2285","article-title":"Interactive learning from policy-dependent human feedback","volume-title":"Proceedings of the 34th International Conference on Machine Learning","author":"MacGlashan"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2022.XVIII.065"},{"key":"ref13","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Rafailov","year":"2023"},{"key":"ref14","first-page":"259","article-title":"PREDILECT: Preferences delineated with zero-shot language-based reasoning in reinforcement learning","volume-title":"Proceedings of the 2024 ACM\/IEEE International Conference on Human-Robot Interaction, HRI \u201924","author":"Holk"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.4324\/9781003022725-3"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3466819"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519735"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/HRI61500.2025.10974241"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v25i1.7979"},{"key":"ref20","article-title":"Gymnasium: A standard interface for reinforcement learning environments","author":"Towers"},{"key":"ref21","article-title":"Introduction to q-learning with OpenAI gym","author":"Tostaeva"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"ref23","article-title":"Garage: A toolkit for reproducible reinforcement learning research","year":"2019"},{"key":"ref24","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"International conference on machine learning","author":"Haarnoja"},{"key":"ref25","first-page":"1094","article-title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","volume-title":"Proceedings of the Conference on Robot Learning","volume":"100","author":"Yu"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s11135-021-01182-y"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1162\/coli.07-034-R2"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/HRI53351.2022.9889368"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/2157689.2157693"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/HRI61500.2025.10974151"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3643834.3660737"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511808418"}],"event":{"name":"2025 34th IEEE International Conference on Robot and Human Interactive Communication (RO-MAN)","location":"Eindhoven, Netherlands","start":{"date-parts":[[2025,8,25]]},"end":{"date-parts":[[2025,8,29]]}},"container-title":["2025 34th IEEE International Conference on Robot and Human Interactive Communication (RO-MAN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11217544\/11217526\/11217743.pdf?arnumber=11217743","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T06:10:51Z","timestamp":1762236651000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11217743\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,25]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/ro-man63969.2025.11217743","relation":{},"subject":[],"published":{"date-parts":[[2025,8,25]]}}}