{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T01:18:39Z","timestamp":1784164719679,"version":"3.55.0"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["2214177"],"award-info":[{"award-number":["2214177"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000181","name":"AFOSR","doi-asserted-by":"publisher","award":["FA9550-22-1-0249"],"award-info":[{"award-number":["FA9550-22-1-0249"]}],"id":[{"id":"10.13039\/100000181","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000006","name":"ONR MURI","doi-asserted-by":"publisher","award":["N00014-22-1-2740"],"award-info":[{"award-number":["N00014-22-1-2740"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000183","name":"ARO","doi-asserted-by":"publisher","award":["W911NF-23-1-0034"],"award-info":[{"award-number":["W911NF-23-1-0034"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006919","name":"MIT Quest for Intelligence","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006919","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007065","name":"NVIDIA Research","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100007065","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/icra55743.2025.11128705","type":"proceedings-article","created":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T17:28:56Z","timestamp":1756834136000},"page":"16847-16853","source":"Crossref","is-referenced-by-count":21,"title":["Guiding Long-Horizon Task and Motion Planning with Vision Language Models"],"prefix":"10.1109","author":[{"given":"Zhutian","family":"Yang","sequence":"first","affiliation":[{"name":"Massachusetts Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Caelan","family":"Garrett","sequence":"additional","affiliation":[{"name":"NVIDIA Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dieter","family":"Fox","sequence":"additional","affiliation":[{"name":"NVIDIA Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tom\u00e1s","family":"Lozano-P\u00e9rez","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Leslie Pack","family":"Kaelbling","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"On the opportunities and risks of foundation models","author":"Bommasani","year":"2021","journal-title":"arXiv preprint"},{"key":"ref2","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023","journal-title":"arXiv preprint"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-091420-084139"},{"key":"ref4","article-title":"Llm+ p: Empowering large language models with optimal planning proficiency","author":"Liu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10342169"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611163"},{"key":"ref7","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022","journal-title":"arXiv preprint"},{"key":"ref8","article-title":"Grounded decoding: Guiding text generation with grounded models for robot control","author":"Huang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"ref11","article-title":"Voxposer: Composable 3d value maps for robotic manipulation with language models","author":"Huang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"Grounding classical task planners via vision-language models","author":"Zhang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref13","article-title":"Replan: Robotic replanning with perception and language models","author":"Skreta","year":"2024","journal-title":"arXiv preprint"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/iros58592.2024.10801328"},{"key":"ref15","article-title":"Copal: Corrective planning of robot actions with large language models","author":"Joublin","year":"2023","journal-title":"arXiv preprint"},{"key":"ref16","article-title":"Doremi: Grounding language model by detecting and recovering from plan-execution misalignment","author":"Guo","year":"2023","journal-title":"arXiv preprint"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-023-10131-7"},{"key":"ref18","article-title":"Translating natural language to planning goals with large-language models","author":"Xie","year":"2023","journal-title":"arXiv preprint"},{"key":"ref19","volume-title":"Pybullet, a python module for physics simulation for games, robotics and machine learning","author":"Coumans","year":"2016"},{"key":"ref20","article-title":"PDDL: The Planning Domain Definition Language","author":"McDermott","year":"1998","journal-title":"Yale Center for Computational Vision and Control, Tech. Rep."},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/icaps.v30i1.6739"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2023.XIX.061"}],"event":{"name":"2025 IEEE International Conference on Robotics and Automation (ICRA)","location":"Atlanta, GA, USA","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,23]]}},"container-title":["2025 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11127273\/11127223\/11128705.pdf?arnumber=11128705","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T06:39:41Z","timestamp":1756881581000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11128705\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/icra55743.2025.11128705","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}