{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:16:44Z","timestamp":1765520204866,"version":"3.48.0"},"reference-count":44,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000161","name":"National Institute of Standards and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100000161","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11246918","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"13452-13459","source":"Crossref","is-referenced-by-count":0,"title":["Real-World Offline Reinforcement Learning from Vision Language Model Feedback"],"prefix":"10.1109","author":[{"given":"Sreyas","family":"Venkataraman","sequence":"first","affiliation":[{"name":"Indian Institute of Technology,Kharagpur"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yufei","family":"Wang","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Kharagpur"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University,IIIS"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Navin Sriram","family":"Ravie","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Kharagpur"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zackory","family":"Erickson","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Kharagpur"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Held","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University,Robotics Institute"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Rl-vlm-f: Reinforcement learning from vision language foundation model feedback","year":"2024","author":"Wang","key":"ref1"},{"article-title":"Offline reinforcement learning with implicit q-learning","year":"2021","author":"Kostrikov","key":"ref2"},{"key":"ref3","first-page":"1179","article-title":"Conservative q-learning for offline reinforcement learning","volume":"33","author":"Kumar","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","year":"2020","author":"Levine","key":"ref4"},{"key":"ref5","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume":"34","author":"Fujimoto","year":"2021","journal-title":"Advances in neural information processing systems"},{"article-title":"Text2reward: Reward shaping with language models for reinforcement learning","year":"2024","author":"Xie","key":"ref6"},{"article-title":"Robogen: Towards unleashing infinite data for automated robot learning via generative simulation","volume-title":"International conference on machine learning","author":"Wang","key":"ref7"},{"article-title":"Eureka: Human-level reward design via coding large language models","year":"2023","author":"Ma","key":"ref8"},{"article-title":"Language to rewards for robotic skill synthesis","year":"2023","author":"Yu","key":"ref9"},{"article-title":"Motif: Intrinsic motivation from artificial intelligence feedback","year":"2023","author":"Klissarov","key":"ref10"},{"key":"ref11","article-title":"Roboclip: One demonstration is enough to learn robot policies","volume":"36","author":"Sontakke","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Vision-language models are zero-shot reward models for reinforcement learning","year":"2023","author":"Rocamonde","key":"ref12"},{"key":"ref13","first-page":"23301","article-title":"Liv: Language-image representations and rewards for robotic control","volume-title":"International Conference on Machine Learning","author":"Ma"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/279943.279964"},{"key":"ref15","first-page":"663","article-title":"Algorithms for inverse reinforcement learning","volume-title":"Proceedings of the Seventeenth International Conference on Machine Learning, ICML \u201900","author":"Ng"},{"key":"ref16","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"8","author":"Ziebart","year":"2008","journal-title":"Aaai"},{"key":"ref17","article-title":"Generative adversarial imitation learning","volume":"29","author":"Ho","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref18","first-page":"49","article-title":"Guided cost learning: Deep inverse optimal control via policy optimization","volume-title":"International conference on machine learning","author":"Finn"},{"article-title":"Learning robust rewards with adversarial inverse reinforcement learning","year":"2017","author":"Fu","key":"ref19"},{"key":"ref20","first-page":"529","article-title":"f-irl: Inverse reinforcement learning via state marginal matching","volume-title":"Conference on Robot Learning","author":"Ni"},{"article-title":"A connection between generative adversarial networks, inverse reinforcement learning, and energy-based models","year":"2016","author":"Finn","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1991.3.1.88"},{"key":"ref23","first-page":"158","article-title":"Implicit behavioral cloning","volume-title":"Conference on Robot Learning","author":"Florence"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.026"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2024.xx.067"},{"key":"ref26","first-page":"1038","article-title":"Toolflownet: Robotic manipulation with tools via predicting tool flow from point clouds","volume-title":"Conference on Robot Learning","author":"Seita"},{"key":"ref27","first-page":"22955","article-title":"Behavior transformers: Cloning k modes with one stone","volume":"35","author":"Shafiullah","year":"2022","journal-title":"Advances in neural information processing systems"},{"key":"ref28","first-page":"15281","article-title":"Unpacking reward shaping: Understanding the benefits of reward engineering on sample complexity","volume":"35","author":"Gupta","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.008"},{"volume-title":"Reinforcement Learning: An Introduction","year":"2018","author":"Sutton","key":"ref31"},{"article-title":"Deep reinforcement learning from human preferences","year":"2023","author":"Christiano","key":"ref32"},{"article-title":"Pebble: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training","year":"2021","author":"Lee","key":"ref33"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"article-title":"Advantage-weighted regression: Simple and scalable off-policy reinforcement learning","year":"2019","author":"Peng","key":"ref35"},{"key":"ref36","article-title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","volume":"30","author":"Qi","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1983.6313077"},{"article-title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","year":"2021","author":"Yu","key":"ref39"},{"article-title":"Softgym: Benchmarking deep reinforcement learning for deformable object manipulation","year":"2021","author":"Lin","key":"ref40"},{"article-title":"Benchmarks and algorithms for offline preference-based reward learning","year":"2023","author":"Shin","key":"ref41"},{"key":"ref42","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"article-title":"D4rl: Datasets for deep data-driven reinforcement learning","year":"2020","author":"Fu","key":"ref43"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3375712"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11246918.pdf?arnumber=11246918","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:13:59Z","timestamp":1765520039000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11246918\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":44,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11246918","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}