{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:18:21Z","timestamp":1778080701381,"version":"3.51.4"},"reference-count":59,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11247438","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"20357-20364","source":"Crossref","is-referenced-by-count":1,"title":["EfficientEQA: An Efficient Approach to Open-Vocabulary Embodied Question Answering"],"prefix":"10.1109","author":[{"given":"Kai","family":"Cheng","sequence":"first","affiliation":[{"name":"Purdue University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengyuan","family":"Li","sequence":"additional","affiliation":[{"name":"Purdue University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xingpeng","family":"Sun","sequence":"additional","affiliation":[{"name":"Purdue University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Byung-Cheol","family":"Min","sequence":"additional","affiliation":[{"name":"Purdue University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amrit Singh","family":"Bedi","sequence":"additional","affiliation":[{"name":"University of Central Florida,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aniket","family":"Bera","sequence":"additional","affiliation":[{"name":"Purdue University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"ref3","article-title":"RT-2: Vision-language-action models transfer web knowledge to robotic control","author":"Brohan","year":"2023"},{"key":"ref4","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"ref5","article-title":"Videonavqa: Bridging the gap between visual and embodied question answering","author":"Cangea","year":"2019"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161534"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01370"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00008"},{"key":"ref9","first-page":"53","article-title":"Neural modular control for embodied question answering","volume-title":"Conference on Robot Learning (CoRL)","author":"Das"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10802733"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02219"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/icra57147.2024.10610090"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00430"},{"key":"ref14","article-title":"Cogvlm2: Visual language models for image and video understanding","author":"Hong","year":"2024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00888"},{"key":"ref16","article-title":"3D-LLM: Injecting the 3D world into large language models","author":"Hong","year":"2023"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"ref18","article-title":"An embodied generalist agent in 3D world","author":"Huang","year":"2023"},{"key":"ref19","first-page":"540","article-title":"Voxposer: Composable 3d value maps for robotic manipulation with language models","volume-title":"Proceedings of The 7th Conference on Robot Learning, volume 229 of Proceedings of Machine Learning Research","author":"Huang"},{"key":"ref20","first-page":"29875","article-title":"Diffvl: scaling up soft body manipulation using vision-language driven differentiable physics","volume":"36","author":"Huang","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref21","article-title":"Language models (mostly) know what they know","author":"Kadavath","year":"2022"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2017.06.005"},{"key":"ref23","article-title":"Prismatic vlms: Investigating the design space of visually-conditioned language models","author":"Karamcheti","year":"2024"},{"key":"ref24","article-title":"Semantic uncertainty: Linguistic invariances for uncertainty estimation in natural language generation","author":"Kuhn","year":"2023"},{"key":"ref25","first-page":"9459","article-title":"Retrieval-augmented generation for knowledgeintensive nlp tasks","volume":"33","author":"Lewis","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref26","article-title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","author":"Li","year":"2024"},{"key":"ref27","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"International conference on machine learning","author":"Li"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01710"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"ref30","article-title":"Teaching models to express their uncertainty in words","author":"Lin","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.02484"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3139957"},{"key":"ref34","article-title":"SQA3D: Situated question answering in 3D scenes","author":"Ma","year":"2022"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01560"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00494"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"ref38","year":"2024","journal-title":"Hello gpt-4"},{"key":"ref39","article-title":"Habitat 3.0: A co-habitat for humans, avatars and robots","author":"Puig","year":"2023"},{"key":"ref40","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref41","article-title":"Habitat-Matterport 3D dataset (HM3D): 1000 large-scale 3D environments for embodied AI","author":"Ramakrishnan","year":"2021"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.089"},{"key":"ref43","article-title":"Robots that ask for help: Uncertainty alignment for large language model planners","author":"Ren","year":"2023"},{"key":"ref44","article-title":"Robots that ask for help: Uncertainty alignment for large language model planners","author":"Ren","year":"2023"},{"key":"ref45","first-page":"2953","article-title":"Exploring models and data for image question answering","volume":"28","author":"Ren","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref46","first-page":"2683","article-title":"Navigation with large language models: Semantic guesswork as a heuristic for planning","volume-title":"Conference on Robot Learning","author":"Shah"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.54"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10801932"},{"key":"ref49","article-title":"Gemini: a family of highly capable multimodal models","author":"Anil","year":"2023"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.330"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00682"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2017.05.001"},{"key":"ref53","article-title":"Can llms express their uncertainty? an empirical evaluation of confidence elicitation in llms","volume-title":"The Twelfth International Conference on Learning Representations","author":"Xiong"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CIRA.1997.613851"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610712"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10802709"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00647"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2024.XX.079"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i7.28597"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Hangzhou, China","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11247438.pdf?arnumber=11247438","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T12:38:31Z","timestamp":1766061511000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11247438\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":59,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11247438","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}