{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T12:14:09Z","timestamp":1777032849938,"version":"3.51.4"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031784552","type":"print"},{"value":"9783031784569","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78456-9_25","type":"book-chapter","created":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T11:24:28Z","timestamp":1733138668000},"page":"389-404","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Zero-Shot Object Navigation with Vision-Language Models Reasoning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6448-003X","authenticated-orcid":false,"given":"Congcong","family":"Wen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yisiyuan","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9131-5854","authenticated-orcid":false,"given":"Hao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1092-3275","authenticated-orcid":false,"given":"Yanjia","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7092-7966","authenticated-orcid":false,"given":"Shuaihang","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9119-6114","authenticated-orcid":false,"given":"Yu","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0190-969X","authenticated-orcid":false,"given":"Hui","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7305-1915","authenticated-orcid":false,"given":"Yu-Shen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9427-3883","authenticated-orcid":false,"given":"Yi","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"key":"25_CR1","unstructured":"AI, A.I.f.: Key features, https:\/\/ai2thor.allenai.org\/robothor\/"},{"key":"25_CR2","doi-asserted-by":"crossref","unstructured":"Al-Halah, Z., Ramakrishnan, S.K., Grauman, K.: Zero experience required: Plug & play modular transfer learning for semantic visual navigation (Apr 2022), https:\/\/arxiv.org\/abs\/2202.02440","DOI":"10.1109\/CVPR52688.2022.01652"},{"issue":"109","key":"25_CR3","first-page":"1","volume":"18","author":"SH Bach","year":"2017","unstructured":"Bach, S.H., Broecheler, M., Huang, B., Getoor, L.: Hinge-loss markov random fields and probabilistic soft logic. J. Mach. Learn. Res. 18(109), 1\u201367 (2017)","journal-title":"J. Mach. Learn. Res."},{"key":"25_CR4","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et\u00a0al.: Language models are few-shot learners (Jan 2020), https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/hash\/1457c0d6bfcb4967418bfb8ac142f64a-Abstract.html"},{"key":"25_CR5","unstructured":"Chang, M., Gupta, A., Gupta, S.: Semantic visual navigation by watching youtube videos (Jan 2020), https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/2cd4e8a2ce081c3d7c32c3cde4312ef7-Abstract.html"},{"key":"25_CR6","unstructured":"Chaplot, D.S., Gandhi, D., Gupta, S., Gupta, A., Salakhutdinov, R.: Learning to explore using active neural slam, https:\/\/iclr.cc\/virtual_2020\/poster_HklXn1BKDH.html"},{"key":"25_CR7","doi-asserted-by":"crossref","unstructured":"Chattopadhyay, P., Hoffman, J., Mottaghi, R., Kembhavi, A.: Robustnav: Towards benchmarking robustness in embodied navigation (Jun 2021), https:\/\/arxiv.org\/abs\/2106.04531","DOI":"10.1109\/ICCV48922.2021.01540"},{"key":"25_CR8","unstructured":"Chen, P., Ji, D., Lin, K., Zeng, R., Li, T.H., Tan, M., Gan, C.: Weakly-supervised multi-granularity map learning for vision-and-language navigation (Oct 2022), https:\/\/arxiv.org\/abs\/2210.07506"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Deitke, M., Han, W., Herrasti, A., Kembhavi, A., Kolve, E., Mottaghi, R., Salvador, J., Schwenk, D., VanderBilt, E., Wallingford, M., et\u00a0al.: Robothor: An open simulation-to-real embodied ai platform (Apr 2020), https:\/\/arxiv.org\/abs\/2004.06799","DOI":"10.1109\/CVPR42600.2020.00323"},{"key":"25_CR10","unstructured":"Deitke, M., VanderBilt, E., Herrasti, A., Weihs, L., Ehsani, K., Salvador, J., Han, W., Kolve, E., Kembhavi, A., Mottaghi, R.: Procthor: Large-scale embodied ai using procedural generation (Dec 2022)"},{"key":"25_CR11","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding, https:\/\/aclanthology.org\/N19-1423\/"},{"key":"25_CR12","unstructured":"Dorbala, V.S., Sigurdsson, G., Piramuthu, R., Thomason, J., Sukhatme, G.S.: Clip-nav: Using clip for zero-shot vision-and-language navigation (Nov 2022), https:\/\/arxiv.org\/abs\/2211.16649"},{"key":"25_CR13","doi-asserted-by":"publisher","unstructured":"Gadre, S.Y., Wortsman, M., Ilharco, G., Schmidt, L., Song, S.: Cows on pasture: Baselines and benchmarks for language-driven zero-shot object navigation. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2023). https:\/\/doi.org\/10.1109\/cvpr52729.2023.02219","DOI":"10.1109\/cvpr52729.2023.02219"},{"key":"25_CR14","doi-asserted-by":"crossref","unstructured":"Gervet, T., Chintala, S., Batra, D., Malik, J., Chaplot, D.S.: Navigating to objects in the real world (Dec 2022), https:\/\/arxiv.org\/abs\/2212.00922","DOI":"10.1126\/scirobotics.adf6991"},{"key":"25_CR15","doi-asserted-by":"crossref","unstructured":"Gomez, C., Hernandez, A.C., Barber, R.: Topological frontier-based exploration and map-building using semantic information (Oct 2019), https:\/\/www.mdpi.com\/1424-8220\/19\/20\/4595","DOI":"10.3390\/s19204595"},{"key":"25_CR16","doi-asserted-by":"publisher","DOI":"10.1109\/iccv.2017.322","author":"K He","year":"2017","unstructured":"He, K., Gkioxari, G., Dollar, P., Girshick, R.: Mask r-cnn. IEEE International Conference on Computer Vision (2017). https:\/\/doi.org\/10.1109\/iccv.2017.322","journal-title":"IEEE International Conference on Computer Vision"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Huang, C., Mees, O., Zeng, A., Burgard, W.: Visual language maps for robot navigation. In: IEEE International Conference on Robotics and Automation. pp. 10608\u201310615. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"25_CR18","doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N.: Mdetr - modulated detection for end-to-end multi-modal understanding. IEEE\/CVF International Conference on Computer Vision (2021)","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"25_CR19","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01441","author":"A Khandelwal","year":"2022","unstructured":"Khandelwal, A., Weihs, L., Mottaghi, R., Kembhavi, A.: Simple but effective: Clip embeddings for embodied ai. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01441","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"25_CR20","unstructured":"Kim, W., Son, B., Kim, I.: Vilt: Vision-and-language transformer without convolution or region supervision (Jun 2021), https:\/\/arxiv.org\/abs\/2102.03334"},{"key":"25_CR21","unstructured":"Leong, K.: Reinforcement learning with frontier-based exploration via autonomous environment (Jul 2023), https:\/\/arxiv.org\/abs\/2307.07296"},{"key":"25_CR22","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01069","author":"LH Li","year":"2022","unstructured":"Li, L.H., Zhang, P., Zhang, H., Yang, J., Li, C., Zhong, Y., Wang, L., Yuan, L., Zhang, L., Hwang, J.N., et al.: Grounded language-image pre-training. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01069","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"25_CR23","unstructured":"Majumdar, A., Aggarwal, G., Devnani, B., Hoffman, J., Batra, D.: Zson: Zero-shot object-goal navigation using multimodal goal embeddings (Jun 2022)"},{"key":"25_CR24","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.01509","author":"O Maksymets","year":"2021","unstructured":"Maksymets, O., Cartillier, V., Gokaslan, A., Wijmans, E., Galuba, W., Lee, S., Batra, D.: Thda: Treasure hunt data augmentation for semantic navigation. IEEE\/CVF International Conference on Computer Vision (2021). https:\/\/doi.org\/10.1109\/iccv48922.2021.01509","journal-title":"IEEE\/CVF International Conference on Computer Vision"},{"key":"25_CR25","doi-asserted-by":"publisher","DOI":"10.1109\/iros47612.2022.9981090","author":"L Mezghan","year":"2022","unstructured":"Mezghan, L., Sukhbaatar, S., Lavril, T., Maksymets, O., Batra, D., Bojanowski, P., Alahari, K.: Memory-augmented reinforcement learning for image-goal navigation. IEEE\/RSJ International Conference on Intelligent Robots and Systems (2022). https:\/\/doi.org\/10.1109\/iros47612.2022.9981090","journal-title":"IEEE\/RSJ International Conference on Intelligent Robots and Systems"},{"key":"25_CR26","unstructured":"Min, S.Y., Chaplot, D.S., Ravikumar, P.K., Bisk, Y., Salakhutdinov, R.: Film: Following instructions in language with modular methods (Sep 2023), https:\/\/openreview.net\/forum?id=qI4542Y2s1D"},{"key":"25_CR27","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A., Stone, A., Neumann, M., Weissenborn, D., Dosovitskiy, A., Mahendran, A., Arnab, A., Dehghani, M., Shen, Z., et\u00a0al.: Simple open-vocabulary object detection with vision transformers (Jul 2022), https:\/\/arxiv.org\/abs\/2205.06230","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"25_CR28","unstructured":"OpenAI: Gpt-3.5 technical report (2023)"},{"key":"25_CR29","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10161345","author":"J Park","year":"2023","unstructured":"Park, J., Yoon, T., Hong, J., Yu, Y., Pan, M., Choi, S.: Zero-shot active visual search (zavis): Intelligent object search for robotic assistants. IEEE International Conference on Robotics and Automation (2023). https:\/\/doi.org\/10.1109\/icra48891.2023.10161345","journal-title":"IEEE International Conference on Robotics and Automation"},{"key":"25_CR30","doi-asserted-by":"crossref","unstructured":"Peters, M.E., Neumann, M., Iyyer, M., Gardner, M., Clark, C., Lee, K., Zettlemoyer, L.: Deep contextualized word representations (Mar 2018), https:\/\/arxiv.org\/abs\/1802.05365","DOI":"10.18653\/v1\/N18-1202"},{"key":"25_CR31","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision (Feb 2021), https:\/\/arxiv.org\/abs\/2103.00020"},{"key":"25_CR32","unstructured":"Rae, J.W., Borgeaud, S., Cai, T., Millican, K., Hoffmann, J., Song, F., Aslanides, J., Henderson, S., Ring, R., Young, S., et\u00a0al.: Scaling language models: Methods, analysis & insights from training gopher (Jan 2022), https:\/\/arxiv.org\/abs\/2112.11446"},{"key":"25_CR33","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00511","author":"R Ramrakhya","year":"2022","unstructured":"Ramrakhya, R., Undersander, E., Batra, D., Das, A.: Habitat-web: Learning embodied object-search strategies from human demonstrations at scale. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.00511","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"25_CR34","doi-asserted-by":"crossref","unstructured":"Verbiest, K., Berrabah, S.A., Colon, E.: Autonomous frontier based exploration for mobile robots (Jan 2015), https:\/\/link.springer.com\/chapter\/10.1007\/978-3-319-22873-0_1","DOI":"10.1007\/978-3-319-22873-0_1"},{"key":"25_CR35","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q., Zhou, D.: Chain-of-thought prompting elicits reasoning in large language models (Jan 2023), https:\/\/arxiv.org\/abs\/2201.11903"},{"key":"25_CR36","doi-asserted-by":"publisher","unstructured":"Yamauchi, B.: A frontier-based approach for autonomous exploration. Proceedings IEEE International Symposium on Computational Intelligence in Robotics and Automation. https:\/\/doi.org\/10.1109\/cira.1997.613851","DOI":"10.1109\/cira.1997.613851"},{"key":"25_CR37","unstructured":"Yao, S., Yu, D., Zhao, J., Shafran, I., Griffiths, T.L., Cao, Y., Narasimhan, K.: Tree of thoughts: Deliberate problem solving with large language models (May 2023), https:\/\/arxiv.org\/abs\/2305.10601"},{"key":"25_CR38","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.01581","author":"J Ye","year":"2021","unstructured":"Ye, J., Batra, D., Das, A., Wijmans, E.: Auxiliary tasks and exploration enable objectgoal navigation. IEEE\/CVF International Conference on Computer Vision (2021). https:\/\/doi.org\/10.1109\/iccv48922.2021.01581","journal-title":"IEEE\/CVF International Conference on Computer Vision"},{"key":"25_CR39","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10161059","author":"B Yu","year":"2023","unstructured":"Yu, B., Kasaei, H., Cao, M.: Frontier semantic exploration for visual target navigation. IEEE International Conference on Robotics and Automation (2023). https:\/\/doi.org\/10.1109\/icra48891.2023.10161059","journal-title":"IEEE International Conference on Robotics and Automation"},{"key":"25_CR40","doi-asserted-by":"crossref","unstructured":"Yu, B., Kasaei, H., Cao, M.: L3mvn: Leveraging large language models for visual target navigation (Apr 2023), https:\/\/arxiv.org\/abs\/2304.05501","DOI":"10.1109\/IROS55552.2023.10342512"},{"key":"25_CR41","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10161289","author":"Q Zhao","year":"2023","unstructured":"Zhao, Q., Zhang, L., He, B., Qiao, H., Liu, Z.: Zero-shot object goal visual navigation. IEEE International Conference on Robotics and Automation (2023). https:\/\/doi.org\/10.1109\/icra48891.2023.10161289","journal-title":"IEEE International Conference on Robotics and Automation"},{"key":"25_CR42","unstructured":"Zheng, K., Zhou, K., Gu, J., Fan, Y., Wang, J., Di, Z., He, X., Wang, X.E.: Jarvis: A neuro-symbolic commonsense reasoning framework for conversational embodied agents (Sep 2022), https:\/\/arxiv.org\/abs\/2208.13266"},{"key":"25_CR43","unstructured":"Zhou, K., Zheng, K., Pryor, C., Shen, Y., Jin, H., Getoor, L., Wang, X.E.: Esc: Exploration with soft commonsense constraints for zero-shot object navigation (Jul 2023), https:\/\/arxiv.org\/abs\/2301.13166"},{"key":"25_CR44","doi-asserted-by":"publisher","DOI":"10.1109\/icra.2017.7989381","author":"Y Zhu","year":"2017","unstructured":"Zhu, Y., Mottaghi, R., Kolve, E., Lim, J.J., Gupta, A., Fei-Fei, L., Farhadi, A.: Target-driven visual navigation in indoor scenes using deep reinforcement learning. IEEE International Conference on Robotics and Automation (2017). https:\/\/doi.org\/10.1109\/icra.2017.7989381","journal-title":"IEEE International Conference on Robotics and Automation"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78456-9_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T12:13:51Z","timestamp":1733141631000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78456-9_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"ISBN":["9783031784552","9783031784569"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78456-9_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"3 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}