{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T21:19:34Z","timestamp":1771708774482,"version":"3.50.1"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100010418","name":"IITP","doi-asserted-by":"publisher","award":["RS-2021-II212068-AIHub\/10%,RS-2021-II211343-GSAI\/15%,2022-0-00951-LBA\/15%,2022-0-00953-PICA\/20%"],"award-info":[{"award-number":["RS-2021-II212068-AIHub\/10%,RS-2021-II211343-GSAI\/15%,2022-0-00951-LBA\/15%,2022-0-00953-PICA\/20%"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"NRF","doi-asserted-by":"publisher","award":["RS-2024-00353991\/20%,RS-2023-00274280\/10%"],"award-info":[{"award-number":["RS-2024-00353991\/20%,RS-2023-00274280\/10%"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003662","name":"KEIT","doi-asserted-by":"publisher","award":["RS-2024-00423940\/10%"],"award-info":[{"award-number":["RS-2024-00423940\/10%"]}],"id":[{"id":"10.13039\/501100003662","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/icra55743.2025.11128677","type":"proceedings-article","created":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T17:28:56Z","timestamp":1756834136000},"page":"16217-16224","source":"Crossref","is-referenced-by-count":3,"title":["Socratic Planner: Self-QA-Based Zero-Shot Planning for Embodied Instruction Following"],"prefix":"10.1109","author":[{"given":"Suyeon","family":"Shin","sequence":"first","affiliation":[{"name":"AI Institute, Seoul National University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sujin","family":"Jeon","sequence":"additional","affiliation":[{"name":"AI Institute, Seoul National University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junghyun","family":"Kim","sequence":"additional","affiliation":[{"name":"AI Institute, Seoul National University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gi-Cheon","family":"Kang","sequence":"additional","affiliation":[{"name":"AI Institute, Seoul National University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Byoung-Tak","family":"Zhang","sequence":"additional","affiliation":[{"name":"AI Institute, Seoul National University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01075"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01004"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01564"},{"key":"ref4","article-title":"FILM: Following instructions in language with modular methods","volume-title":"International Conference on Learning Representations","author":"Min","year":"2022"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.368"},{"key":"ref6","first-page":"706","article-title":"A persistent spatial semantic representation for high-level natural language instruction execution","volume-title":"Conference on Robot Learning","author":"Blukis","year":"2022"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3178804"},{"key":"ref8","article-title":"Prompter: Utilizing large language model prompting for a data efficient embodied instruction following","author":"Inoue","year":"2022","journal-title":"arXiv preprint"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01504"},{"key":"ref10","article-title":"Lebp-language expectation & binding policy: A two-stream framework for embodied vision-and-language interaction task learning agents","author":"Liu","year":"2022","journal-title":"arXiv preprint"},{"key":"ref11","article-title":"Are we there yet? learning to localize in embodied instruction following","author":"Storks","year":"2021","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"Agent with the big picture: Perceiving surroundings for interactive instruction following","volume-title":"Embodied AI Workshop CVPR","author":"Kim","year":"2021"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/128"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00190"},{"key":"ref15","article-title":"Embodied bert: A transformer model for embodied, language-guided visual task completion","author":"Suglia","year":"2021","journal-title":"arXiv preprint"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00280"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"ref18","article-title":"Neuro-symbolic procedural planning with commonsense prompting","volume-title":"The Eleventh International Conference on Learning Representations","author":"Lu","year":"2023"},{"key":"ref19","article-title":"Jarvis: A neuro-symbolic commonsense reasoning framework for conversational embodied agents","author":"Zheng","year":"2022","journal-title":"arXiv preprint"},{"key":"ref20","article-title":"Inner monologue: Embodied reasoning through planning with language models","volume-title":"6th Annual Conference on Robot Learning","author":"Huang","year":"2022"},{"key":"ref21","article-title":"Embodied task planning with large language models","author":"Wu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref22","article-title":"React: Synergizing reasoning and acting in language models","volume-title":"The Eleventh International Conference on Learning Representations","author":"Yao","year":"2023"},{"key":"ref23","article-title":"Thinkbot: Embodied instruction following with thought chain reasoning","author":"Lu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref24","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022","journal-title":"arXiv preprint"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1515\/9781438419329"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00430"},{"key":"ref27","article-title":"A hierarchical attention model for action learning from realistic environments and directives","volume-title":"European Conference on Computer Vision (ECCV) EVAL Workshop","author":"Van-Quang Nguyen","year":"2020"},{"key":"ref28","first-page":"9118","article-title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","volume-title":"International Conference on Machine Learning","author":"Huang","year":"2022"},{"key":"ref29","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"35","author":"Wei","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref30","article-title":"Least-to-most prompting enables complex reasoning in large language models","volume-title":"The Eleventh International Conference on Learning Representations","author":"Zhou","year":"2023"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.441"},{"key":"ref32","article-title":"Measuring and narrowing the compositionality gap in language models","author":"Press","year":"2022","journal-title":"arXiv preprint"},{"key":"ref33","article-title":"Lota-bench: Benchmarking language-oriented task planners for embodied agents","volume-title":"The Twelfth International Conference on Learning Representations","author":"Choi","year":"2024"},{"key":"ref34","article-title":"Planning with large language models via corrective re-prompting","volume-title":"NeurIPS 2022 Foundation Models for Decision Making Workshop","author":"Raman","year":"2022"},{"key":"ref35","article-title":"Gpt-4o: Visual perception performance of multimodal large language models in piglet activity understanding","author":"Wu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref36","first-page":"27 730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref37","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023","journal-title":"arXiv preprint"},{"key":"ref38","article-title":"Ai2-thor: An interactive 3d environment for visual ai","author":"Kolve","year":"2017","journal-title":"arXiv preprint"},{"key":"ref39","first-page":"22199","article-title":"Large language models are zero-shot reasoners","volume":"35","author":"Kojima","year":"2022","journal-title":"Advances in neural information processing systems"}],"event":{"name":"2025 IEEE International Conference on Robotics and Automation (ICRA)","location":"Atlanta, GA, USA","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,23]]}},"container-title":["2025 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11127273\/11127223\/11128677.pdf?arnumber=11128677","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T06:47:09Z","timestamp":1756882029000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11128677\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/icra55743.2025.11128677","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}