{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T20:28:43Z","timestamp":1785356923196,"version":"3.55.0"},"reference-count":42,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1109\/lra.2025.3589162","type":"journal-article","created":{"date-parts":[[2025,7,14]],"date-time":"2025-07-14T17:44:18Z","timestamp":1752515058000},"page":"8850-8857","source":"Crossref","is-referenced-by-count":3,"title":["Towards Autonomous Reinforcement Learning for Real-World Robotic Manipulation With Large Language Models"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2995-4110","authenticated-orcid":false,"given":"Niccol\u00f2","family":"Turcato","sequence":"first","affiliation":[{"name":"Department of Information Engineering, University of Padova, Padova, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6119-6399","authenticated-orcid":false,"given":"Matteo","family":"Iovino","sequence":"additional","affiliation":[{"name":"ABB Corporate Research, V&#x00E4;ster&#x00E5;s, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aris","family":"Synodinos","sequence":"additional","affiliation":[{"name":"ABB Corporate Research, V&#x00E4;ster&#x00E5;s, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7656-7797","authenticated-orcid":false,"given":"Alberto","family":"Dalla Libera","sequence":"additional","affiliation":[{"name":"Department of Information Engineering, University of Padova, Padova, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6506-5898","authenticated-orcid":false,"given":"Ruggero","family":"Carli","sequence":"additional","affiliation":[{"name":"Department of Information Engineering, University of Padova, Padova, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1133-0884","authenticated-orcid":false,"given":"Pietro","family":"Falco","sequence":"additional","affiliation":[{"name":"Department of Information Engineering, University of Padova, Padova, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"A survey on large language models: Applications, challenges, limitations, and practical usage","author":"Hadi","year":"2023","journal-title":"Authorea Preprints"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.129963"},{"key":"ref3","first-page":"287","article-title":"Do as I can, not as I say: Grounding language in robotic affordances","volume-title":"Proc. Conf. Robot Learn.","author":"Brohan","year":"2023"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"ref5","article-title":"Plan-seq-learn: Language model guided RL for solving long horizon robotics tasks","volume-title":"Proc. NeurIPS 2023 Found. Models Decision Making Workshop","author":"Dalal","year":"2023"},{"key":"ref6","article-title":"Automatic behavior tree expansion with LLMs for robotic manipulation","volume-title":"Proc. 2025 IEEE Int. Conf. Robot. Autom.","author":"Styrud","year":"2025"},{"key":"ref7","first-page":"374","article-title":"Language to rewards for robotic skill synthesis","volume-title":"Proc. 7th Conf. Robot Learn.","volume":"229,","author":"Yu","year":"2023"},{"key":"ref8","first-page":"2165","article-title":"RT-2: Vision-language-action models transfer web knowledge to robotic control","volume-title":"Proc. Conf. Robot Learn.","author":"Zitkovich","year":"2023"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3410155"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3207346"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"issue":"2","key":"ref12","first-page":"239","article-title":"Overview of deep reinforcement learning improvements and applications","volume":"22","author":"Zhang","year":"2021","journal-title":"J. Internet Technol."},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-41188-6_3"},{"key":"ref14","article-title":"GPT-4 technical report","author":"Achiam","year":"2023"},{"key":"ref15","article-title":"Eureka: Human-level reward design via coding large language models","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Ma","year":"2024"},{"key":"ref16","first-page":"1303","article-title":"Learning language-conditioned robot behavior from offline data and crowd-sourced annotation","volume-title":"Proc. 5th Conf. Robot Learn.","volume":"164","author":"Nair","year":"2022"},{"key":"ref17","first-page":"51936","article-title":"RoboGen: Towards unleashing infinite data for automated robot learning via generative simulation","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","volume":"235","author":"Wang","year":"2024"},{"key":"ref18","article-title":"Self-eefined large language model as automated reward function designer for deep reinforcement learning in robotics","author":"Song","year":"2023"},{"key":"ref19","article-title":"Text2Reward: Reward shaping with language models for reinforcement learning","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Xie","year":"2024"},{"key":"ref20","article-title":"CurricuLLM: Automatic task curricula design for learning complex robot skills using large language models","author":"Ryu","year":"2024"},{"key":"ref21","article-title":"AnyBipe: An end-to-end framework for training and deploying bipedal robots guided by large language models","author":"Yao","year":"2024"},{"key":"ref22","first-page":"8657","article-title":"Guiding pretraining in reinforcement learning with large language models","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Du","year":"2023"},{"key":"ref23","article-title":"REvolve: Reward evolution with large language models using human feedback","volume-title":"Proc. 13th Int. Conf. Learn. Representations","author":"Hazra","year":"2025"},{"key":"ref24","article-title":"ExploRLLM: Guiding exploration in reinforcement learning with large language models","volume-title":"Proc. RSS 2024 Workshop: Data Gener. Robot.","author":"Ma","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3357432"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3400189"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610421"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3528663"},{"key":"ref29","article-title":"Large language models as generalizable policies for embodied tasks","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"Szot","year":"2024"},{"key":"ref30","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2018"},{"key":"ref31","first-page":"1481","article-title":"Policy learning in SE (3) action spaces","volume-title":"Proc. Conf. Robot Learn.","author":"Wang","year":"2021"},{"key":"ref32","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto","year":"2018"},{"key":"ref33","first-page":"62244","article-title":"Cal-QL: Calibrated offline RL pre-training for efficient online fine-tuning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Nakamoto","year":"2023"},{"key":"ref34","first-page":"6152","article-title":"PEBBLE: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lee","year":"2021"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.094"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3250269"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/s00170-021-07682-3"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"ref40","article-title":"Soft actor-critic algorithms and applications","author":"Haarnoja","year":"2018"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.3028529"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/icra.2015.7138994"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7083369\/11082640\/11080043.pdf?arnumber=11080043","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T17:57:29Z","timestamp":1753379849000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11080043\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9]]},"references-count":42,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/lra.2025.3589162","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9]]}}}