{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T20:02:41Z","timestamp":1784923361292,"version":"3.55.0"},"reference-count":57,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["W2411052"],"award-info":[{"award-number":["W2411052"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Young Taishan Scholars Program of Shandong Province","award":["tsqn202408072"],"award-info":[{"award-number":["tsqn202408072"]}]},{"name":"Outstanding Youth Foundation of Shanxi Province","award":["2025JC-JCQN-092"],"award-info":[{"award-number":["2025JC-JCQN-092"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Robot."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tro.2026.3710412","type":"journal-article","created":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T19:57:57Z","timestamp":1783367877000},"page":"2877-2897","source":"Crossref","is-referenced-by-count":0,"title":["Generative Adversarial Self-Imitation Learning With Large Language Model Feedback for Robot Control and Navigation"],"prefix":"10.1109","volume":"42","author":[{"given":"Ke","family":"Zhang","sequence":"first","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9490-307X","authenticated-orcid":false,"given":"Zheng","family":"Fang","sequence":"additional","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Enqi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zicheng","family":"Sun","sequence":"additional","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0300-6892","authenticated-orcid":false,"given":"Jianwu","family":"Fang","sequence":"additional","affiliation":[{"name":"Xi\u2019an Jiaotong University","place":["Xi\u2019an, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1570-6797","authenticated-orcid":false,"given":"Jie","family":"Huang","sequence":"additional","affiliation":[{"name":"Civil Aviation Chengdu Information Technology Company, Ltd.","place":["Chengdu, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0734-6621","authenticated-orcid":false,"given":"Eric","family":"Nichols","sequence":"additional","affiliation":[{"name":"Honda Research Institute Japan Company, Ltd.","place":["Wako, Japan"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3191-6818","authenticated-orcid":false,"given":"Randy","family":"Gomez","sequence":"additional","affiliation":[{"name":"Honda Research Institute Japan Company, Ltd.","place":["Wako, Japan"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6826-4721","authenticated-orcid":false,"given":"Bo","family":"He","sequence":"additional","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4994-9343","authenticated-orcid":false,"given":"Jianru","family":"Xue","sequence":"additional","affiliation":[{"name":"Xi\u2019an Jiaotong University","place":["Xi\u2019an, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1728-5711","authenticated-orcid":false,"given":"Guangliang","family":"Li","sequence":"additional","affiliation":[{"name":"Ocean University of China","place":["Qingdao, China"]}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i27.35095"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref4","article-title":"Mastering the real-time strategy game starcraft II","author":"DeepMind","year":"2019"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2023.3262461"},{"key":"ref6","first-page":"308","article-title":"Fully autonomous real-world reinforcement learning with applications to mobile manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Sun","year":"2022"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-05732-2"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-06419-4"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.adi9579"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.adi8022"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-024-00891-x"},{"key":"ref12","first-page":"4344","article-title":"Learning by playing solving sparse reward tasks from scratch","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Riedmiller","year":"2018"},{"key":"ref13","first-page":"661","article-title":"Efficient reductions for imitation learning","volume-title":"Proc. 30th Int. Conf. Artif. Intell. Statist.","author":"Ross","year":"2010"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2024.3372778"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8463189"},{"key":"ref17","article-title":"Algorithms for inverse reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"1","author":"Ng","year":"2000"},{"key":"ref18","first-page":"2760","article-title":"Model-free imitation learning with policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ho","year":"2016"},{"key":"ref19","first-page":"4565","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho","year":"2016"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9248"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969125"},{"key":"ref22","article-title":"Generative adversarial self-imitation learning","author":"Guo","year":"2018"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160939"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3550743"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3400189"},{"key":"ref26","first-page":"287","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","volume-title":"Proc. Conf. Robot Learn.","author":"Brohan","year":"2023"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.65109\/vwdm2269"},{"key":"ref28","first-page":"26516","article-title":"Eureka: Human-level reward design via coding large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ma","year":"2024"},{"key":"ref29","first-page":"1","article-title":"Text2Reward: Automated dense reward function generation for reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Xie","year":"2024"},{"key":"ref30","first-page":"28446","article-title":"Vision-language models are zero-shot reward models for reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Rocamonde","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611483"},{"key":"ref32","first-page":"1","article-title":"Enabling intelligent interactions between an agent and an LLM: A reinforcement learning approach","volume-title":"Proc. Reinforcement Learn. Conf.","author":"Hu","year":"2024"},{"key":"ref33","first-page":"1","article-title":"Towards principled methods for training generative adversarial networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Arjovsky","year":"2017"},{"key":"ref34","first-page":"5326","article-title":"Robust imitation of diverse behaviors","volume-title":"Proc. 31st Int. Conf. Neural Inf. Process. Syst.","author":"Wang","year":"2017"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10801451"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611642"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160374"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3546860"},{"key":"ref39","first-page":"1071","article-title":"Out-of-dynamics imitation learning from multimodal demonstrations","volume-title":"Proc. 6th Conf. Robot Learn.","volume":"205","author":"Qiu","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10801662"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610371"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3319502.3374832"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3451360"},{"key":"ref44","article-title":"Self-imitation learning from demonstrations","author":"Pshikhachev","year":"2022"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3196122"},{"key":"ref46","first-page":"3676","article-title":"Grounding large language models in interactive environments with online reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Carta","year":"2023"},{"key":"ref47","first-page":"4033","article-title":"Teaching robots with show and tell: Using foundation models to synthesize robot policies from language and visual demonstration","volume-title":"Proc. 8th Annu. Conf. Robot Learn.","author":"Murray","year":"2024"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610744"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3265893"},{"key":"ref50","first-page":"1","article-title":"Plan-seq-learn: Language model guided RL for solving long horizon robotics tasks","volume-title":"Proc. Workshop Large Lang. Model Agents","author":"Dalal","year":"2024"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3511402"},{"key":"ref52","first-page":"1643","article-title":"HYPERmotion: Learning hybrid behavior planning for autonomous loco-manipulation","volume-title":"Proc. 8th Annu. Conf. Robot Learn.","author":"Wang","year":"2024"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610948"},{"key":"ref54","first-page":"3766","article-title":"Scaling up and distilling down: Language-guided robot skill acquisition","volume-title":"Proc. Annu. Conf. Robot Learn.","author":"Ha","year":"2023"},{"key":"ref55","first-page":"8657","article-title":"Guiding pretraining in reinforcement learning with large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Du","year":"2023"},{"key":"ref56","first-page":"1","article-title":"Accelerating reinforcement learning of robotic manipulations via feedback from large language models","volume-title":"7th Conf. Robot Learn. Workshop","author":"Chu","year":"2024"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"}],"container-title":["IEEE Transactions on Robotics"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/8860\/11297026\/11595563.pdf?arnumber=11595563","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T19:05:09Z","timestamp":1784919909000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11595563\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":57,"URL":"https:\/\/doi.org\/10.1109\/tro.2026.3710412","relation":{},"ISSN":["1552-3098","1941-0468"],"issn-type":[{"value":"1552-3098","type":"print"},{"value":"1941-0468","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}