{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T20:29:47Z","timestamp":1785356987656,"version":"3.55.0"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1109\/lra.2024.3477090","type":"journal-article","created":{"date-parts":[[2024,10,9]],"date-time":"2024-10-09T17:56:54Z","timestamp":1728496614000},"page":"10567-10574","source":"Crossref","is-referenced-by-count":71,"title":["GPT-4V(ision) for Robotics: Multimodal Task Planning From Human Demonstration"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8278-2373","authenticated-orcid":false,"given":"Naoki","family":"Wake","sequence":"first","affiliation":[{"name":"Applied Robotics Research, Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0157-7500","authenticated-orcid":false,"given":"Atsushi","family":"Kanehira","sequence":"additional","affiliation":[{"name":"Applied Robotics Research, Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5408-3089","authenticated-orcid":false,"given":"Kazuhiro","family":"Sasabuchi","sequence":"additional","affiliation":[{"name":"Applied Robotics Research, Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7457-2878","authenticated-orcid":false,"given":"Jun","family":"Takamatsu","sequence":"additional","affiliation":[{"name":"Applied Robotics Research, Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9758-9357","authenticated-orcid":false,"given":"Katsushi","family":"Ikeuchi","sequence":"additional","affiliation":[{"name":"Applied Robotics Research, Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Vima: General robot manipulation with multimodal prompts","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Jiang","year":"2023"},{"key":"ref2","article-title":"RT-2: Vision-language-action models transfer web knowledge to robotic control","author":"Brohan","year":"2023"},{"key":"ref3","article-title":"RT-1: Robotics transformer for real-world control at scale","author":"Brohan","year":"2022"},{"key":"ref4","article-title":"Vision-language foundation models as effective robot imitators","author":"Li","year":"2023"},{"key":"ref5","article-title":"Do as I can, not as I say: Grounding language in robotic affordances","author":"Ahn","year":"2022"},{"key":"ref6","first-page":"2663","article-title":"Mutex: Learning unified policies from multimodal task specifications","volume-title":"Proc. 7th Conf. Robot Learn.","volume":"229","author":"Shah","year":"69, 2023"},{"key":"ref7","article-title":"Mastering robot manipulation with multimodal prompts through pretraining and multi-task fine-tuning","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Li","year":"2024"},{"key":"ref8","article-title":"Chatgpt","year":"2022"},{"key":"ref9","article-title":"Gpt-4","year":"2023"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3310935"},{"key":"ref11","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Huang","year":"2022"},{"key":"ref12","article-title":"Creative robot tool use with large language models","author":"Xu","year":"2023"},{"key":"ref13","article-title":"Generalizable long-horizon manipulations with large language models","author":"Zhou","year":"2023"},{"key":"ref14","article-title":"Grid: Scene-graph-based instruction-driven robotic task planning","author":"Ni","year":"2023"},{"key":"ref15","article-title":"Interactive task planning with language models","author":"Li","year":"2023"},{"key":"ref16","article-title":"Tree-planner: Efficient close-loop task planning with large language models","author":"Hu","year":"2023"},{"key":"ref17","article-title":"ChatGPT for robotics: Design principles and model abilities","volume":"2","author":"Vemprala","year":"2023","journal-title":"Microsoft Auton. Syst. Robot. Res"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9341289"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636342"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/SII52469.2022.9708836"},{"key":"ref21","first-page":"785","article-title":"Perceiver-actor: A multi-task transformer for robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Shridhar","year":"2023"},{"key":"ref22","first-page":"287","article-title":"Do as I can, not as I say: Grounding language in robotic affordances","volume-title":"Proc. Conf. Robot Learn.","author":"Brohan","year":"2023"},{"key":"ref23","article-title":"Inner monologue: Embodied reasoning through planning with language models","author":"Huang","year":"2022"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10342169"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"ref26","first-page":"7973","article-title":"Learning neuro-symbolic programs for language guided robot manipulation","volume-title":"Proc. IEEE Int. Conf. Robot. Automat.","author":"Namasivayam","year":"2023"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160640"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-023-10133-5"},{"key":"ref29","article-title":"Socratic models: Composing zero-shot multimodal reasoning with language","author":"Zeng","year":"2022"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"ref31","article-title":"Planning with large language models via corrective re-prompting","volume-title":"Proc. NeurIPS Found. Models Decis. Mak. Workshop","author":"Raman","year":"2022"},{"key":"ref32","article-title":"Translating natural language to planning goals with large-language models","author":"Xie","year":"2023"},{"key":"ref33","article-title":"Language conditioned imitation learning over unstructured data","author":"Lynch","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161125"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-023-10131-7"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3265893"},{"key":"ref37","article-title":"Instruction-following agents with multimodal transformer","author":"Liu","year":"2022"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160396"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610948"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-091420-084139"},{"key":"ref41","article-title":"Human-assisted continual robot learning with foundation models","author":"Parakh","year":"2023"},{"key":"ref42","first-page":"14070","article-title":"CAPE: Corrective actions from precondition errors using large language models","volume-title":"Proc. 2nd Workshop Lang. Robot Learn.: Lang. Grounding","author":"Raman","year":"2023"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.4324\/9781315740218"},{"key":"ref45","first-page":"540","article-title":"VoxPoser: Composable 3D value maps for robotic manipulation with language models","volume-title":"Proc. 7th Conf. Robot Learn.","volume":"229","author":"Huang","year":"69, 2023"},{"key":"ref46","article-title":"MOKA: Open-vocabulary robotic manipulation through mark-based visual prompting","author":"Liu","year":"2024"},{"key":"ref47","article-title":"ManipVQA: Injecting robotic affordance and physically grounded information into multi-modal large language models","author":"Huang","year":"2024"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01370"},{"key":"ref49","article-title":"PIVOT: Iterative visual prompting elicits actionable knowledge for VLMs","author":"Nasiriany","year":"2024"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.3389\/fcomp.2024.1235239"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/AIM46323.2023.10196126"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/s00138-023-01408-z"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.3044029"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/IEEECONF49454.2021.9382750"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1177\/02783649231212929"},{"key":"ref56","article-title":"Related systems","volume-title":"Annals of Math. Studies","author":"Kuhn","year":"1956"},{"key":"ref57","first-page":"10377","article-title":"Verbal focus-of-attention system for learning-from-demonstration","volume-title":"Proc. IEEE Int. Conf. Robot. Automat.","author":"Wake","year":"2021"},{"key":"ref58","article-title":"Yolo","year":"2023"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.1996.506914"},{"key":"ref60","article-title":"Task-sequencing simulator: Integrated machine learning to execution simulation for robot manipulation","author":"Sasabuchi","year":"2023"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/Humanoids53995.2022.10000167"},{"key":"ref62","article-title":"Learning-from-observation system considering hardware-level reusability","author":"Takamatsu","year":"2022"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1038\/sdata.2018.101"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2991965"},{"key":"ref66","article-title":"Pythonvideoannotator.","year":"2015"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7083369\/10683798\/10711245.pdf?arnumber=10711245","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T19:38:20Z","timestamp":1734982700000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10711245\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11]]},"references-count":67,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/lra.2024.3477090","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11]]}}}