{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,20]],"date-time":"2025-11-20T19:09:05Z","timestamp":1763665745801,"version":"3.41.0"},"publisher-location":"Cham","reference-count":47,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031919886","type":"print"},{"value":"9783031919893","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91989-3_5","type":"book-chapter","created":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:33:05Z","timestamp":1748194385000},"page":"68-85","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Space3D-Bench: Spatial 3D Question Answering Benchmark"],"prefix":"10.1007","author":[{"given":"Emilia","family":"Szyma\u0144ska","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mihai","family":"Dusmanu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jan-Willem","family":"Buurlage","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mahdi","family":"Rad","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marc","family":"Pollefeys","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"5_CR1","doi-asserted-by":"publisher","unstructured":"Abdin, M.I., et al.: Phi-3 technical report: a highly capable language model locally on your phone. CoRR abs\/2404.14219 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2404.14219","DOI":"10.48550\/ARXIV.2404.14219"},{"key":"5_CR2","doi-asserted-by":"publisher","unstructured":"Anderson, P., et al.: Vision-and-language navigation: interpreting visually-grounded navigation instructions in real environments. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3674\u20133683 (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00387","DOI":"10.1109\/CVPR.2018.00387"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Azuma, D., Miyanishi, T., Kurita, S., Kawanabe, M.: Scanqa: 3d question answering for spatial scene understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022)","DOI":"10.1109\/CVPR52688.2022.01854"},{"key":"5_CR4","doi-asserted-by":"publisher","unstructured":"Bozkir, E., \u00d6zdel, S., Lau, K.H.C., Wang, M., Gao, H., Kasneci, E.: Embedding large language models into extended reality: opportunities and challenges for inclusion, engagement, and privacy. In: ACM Conversational User Interfaces 2024, CUI 2024, ACM, July 2024. https:\/\/doi.org\/10.1145\/3640794.3665563, http:\/\/dx.doi.org\/10.1145\/3640794.3665563","DOI":"10.1145\/3640794.3665563"},{"key":"5_CR5","unstructured":"Brohan, A., et al.: Rt-2: vision-language-action models transfer web knowledge to robotic control (2023). https:\/\/arxiv.org\/abs\/2307.15818"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Caesar, H., et al.: nuscenes: a multimodal dataset for autonomous driving. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11618\u201311628 (2019). https:\/\/api.semanticscholar.org\/CorpusID:85517967","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"5_CR7","unstructured":"Chang, H., et al.: Context-aware entity grounding with open-vocabulary 3d scene graphs. In: 7th Annual Conference on Robot Learning (2023). https:\/\/openreview.net\/forum?id=cjEI5qXoT0"},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Chen, D.Z., Chang, A.X., Nie\u00dfner, M.: Scanrefer: 3d object localization in rgb-d scans using natural language. In: 16th European Conference on Computer Vision (ECCV) (2020)","DOI":"10.1007\/978-3-030-58565-5_13"},{"key":"5_CR9","unstructured":"Cho, J.H., et al.: Language-image models with 3d understanding (2024). https:\/\/arxiv.org\/abs\/2405.03685"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: Scannet: richly-annotated 3d reconstructions of indoor scenes. In: Proceedings of Computer Vision and Pattern Recognition (CVPR), IEEE (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"5_CR11","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Burstein, J., Doran, C., Solorio, T. (eds.) Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics, Minneapolis, Minnesota, June 2019. https:\/\/doi.org\/10.18653\/v1\/N19-1423, https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"5_CR12","unstructured":"Fang, C.M., Zieli\u0144ski, K., Maes, P., Paradiso, J., Blumberg, B., Kj\u00e6rgaard, M.B.: Enabling waypoint generation for collaborative robots using llms and mixed reality (2024) https:\/\/arxiv.org\/abs\/2403.09308"},{"key":"5_CR13","doi-asserted-by":"publisher","unstructured":"Gu, J., Stefani, E., Wu, Q., Thomason, J., Wang, X.: Vision-and-language navigation: a survey of tasks, methods, and future directions. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Association for Computational Linguistics (2022). https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.524, http:\/\/dx.doi.org\/10.18653\/v1\/2022.acl-long.524","DOI":"10.18653\/v1\/2022.acl-long.524"},{"key":"5_CR14","doi-asserted-by":"crossref","unstructured":"Gu, Q., et al.: Conceptgraphs: open-vocabulary 3d scene graphs for perception and planning (2023)","DOI":"10.1109\/ICRA57147.2024.10610243"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Hong, Y., Lin, C., Du, Y., Chen, Z., Tenenbaum, J.B., Gan, C.: 3d concept learning and reasoning from multi-view images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.00888"},{"key":"5_CR16","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"5_CR17","doi-asserted-by":"publisher","unstructured":"Kasneci, E., et al.: Chatgpt for good? on opportunities and challenges of large language models for education. Learn. Individ. Differ. 103, 102274 (2023).https:\/\/doi.org\/10.1016\/j.lindif.2023.102274, https:\/\/www.sciencedirect.com\/science\/article\/pii\/S1041608023000195","DOI":"10.1016\/j.lindif.2023.102274"},{"key":"5_CR18","unstructured":"Lewis, P., et al.: Retrieval-augmented generation for knowledge-intensive nlp tasks. In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol.\u00a033, pp. 9459\u20139474. Curran Associates, Inc. (2020). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/6b493230205f780e1bc26945df7481e5-Paper.pdf"},{"key":"5_CR19","unstructured":"Li, M., et al.: M3dbench: let\u2019s instruct large models with multi-modal 3d prompts (2023)"},{"key":"5_CR20","doi-asserted-by":"publisher","unstructured":"Liu, J.: LlamaIndex, November 2022. https:\/\/doi.org\/10.5281\/zenodo.1234, https:\/\/github.com\/jerryjliu\/llama_index","DOI":"10.5281\/zenodo.1234"},{"key":"5_CR21","unstructured":"Ma, X., et al.: When llms step into the 3d world: a survey and meta-analysis of 3d tasks via multi-modal large language models. In: IEEE (2024)"},{"key":"5_CR22","unstructured":"Ma, X., et al.: Sqa3d: situated question answering in 3d scenes. In: International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=IDJx97BC38"},{"key":"5_CR23","unstructured":"Mao, J., Qian, Y., Ye, J., Zhao, H., Wang, Y.: Gpt-driver: learning to drive with gpt (2023). https:\/\/arxiv.org\/abs\/2310.01415"},{"key":"5_CR24","unstructured":"Mao, J., Ye, J., Qian, Y., Pavone, M., Wang, Y.: A language agent for autonomous driving (2024). https:\/\/arxiv.org\/abs\/2311.10813"},{"key":"5_CR25","unstructured":"Microsoft: Semantic kernel (2024). https:\/\/github.com\/microsoft\/semantic-kernel, accessed: 2024-08-01"},{"key":"5_CR26","unstructured":"Miriam\u00a0Schmidts, Nicholas, M.: Giner.: understanding the basics: introduction to the language of spatial analysis. https:\/\/proceedings.esri.com\/library\/userconf\/proc18\/tech-workshops\/tw_1593-380.pdf, Accessed 29 July 2024"},{"key":"5_CR27","doi-asserted-by":"publisher","unstructured":"Mirzaee, R., Rajaby\u00a0Faghihi, H., Ning, Q., Kordjamshidi, P.: SPARTQA: a textual question answering benchmark for spatial reasoning. In: Toutanova, K., Rumshisky, A., Zettlemoyer, L., Hakkani-Tur, D., Beltagy, I., Bethard, S., Cotterell, R., Chakraborty, T., Zhou, Y. (eds.) Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. pp. 4582\u20134598. Association for Computational Linguistics, June 2021. https:\/\/doi.org\/10.18653\/v1\/2021.naacl-main.364, https:\/\/aclanthology.org\/2021.naacl-main.364","DOI":"10.18653\/v1\/2021.naacl-main.364"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Ning, H., Li, Z., Akinboyewa, T., Lessani, M.N.: Llm-find: an autonomous gis agent framework for geospatial data retrieval (2024). https:\/\/arxiv.org\/abs\/2407.21024","DOI":"10.1080\/17538947.2025.2458688"},{"key":"5_CR29","unstructured":"OpenAI: Gpt-4v(ision) system card (2023). https:\/\/api.semanticscholar.org\/CorpusID:263218031, Accessed 01 Aug 2024"},{"key":"5_CR30","unstructured":"OpenAI: New and improved embedding model (2024). https:\/\/openai.com\/index\/new-and-improved-embedding-model\/, Accessed 01 Aug 2024"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Qi, Z., et al.: Gpt4point: a unified framework for point-language understanding and generation. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.02495"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Qian, T., Chen, J., Zhuo, L., Jiao, Y., Jiang, Y.: Nuscenes-qa: a multi-modal visual question answering benchmark for autonomous driving scenario. In: AAAI Conference on Artificial Intelligence (2023). https:\/\/api.semanticscholar.org\/CorpusID:258866014","DOI":"10.1609\/aaai.v38i5.28253"},{"key":"5_CR33","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR, 18\u201324 July 2021. https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Savva, M., et al.: Habitat: a platform for embodied AI research. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00943"},{"key":"5_CR35","unstructured":"Straub, J., et al.: The replica dataset: a digital replica of indoor spaces. arXiv preprint arXiv:1906.05797 (2019)"},{"key":"5_CR36","unstructured":"Torre, F.D.L., Fang, C.M., Huang, H., Banburski-Fahey, A., Fernandez, J.A., Lanier, J.: Llmr: real-time prompting of interactive worlds using large language models (2024). https:\/\/arxiv.org\/abs\/2309.12276"},{"key":"5_CR37","doi-asserted-by":"crossref","unstructured":"Wald, J., Avetisyan, A., Navab, N., Tombari, F., Niessner, M.: Rio: 3d object instance re-localization in changing indoor environments. In: Proceedings IEEE International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00775"},{"key":"5_CR38","unstructured":"Wang, J., et al.: Large language models for robotics: opportunities, challenges, and perspectives (2024). https:\/\/arxiv.org\/abs\/2401.04334"},{"key":"5_CR39","doi-asserted-by":"crossref","unstructured":"Xu, R., Wang, X., Wang, T., Chen, Y., Pang, J., Lin, D.: Pointllm: empowering large language models to understand point clouds. arXiv preprint arXiv:2308.16911 (2023)","DOI":"10.1007\/978-3-031-72698-9_8"},{"key":"5_CR40","first-page":"1","volume":"01","author":"X Yan","year":"2023","unstructured":"Yan, X., et al.: Comprehensive visual question answering on point clouds through compositional scene manipulation. IEEE Trans. Vis. Comput. Graph. 01, 1\u201313 (2023)","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"issue":"3","key":"5_CR41","doi-asserted-by":"publisher","first-page":"1772","DOI":"10.1109\/TVCG.2022.3225327","volume":"30","author":"S Ye","year":"2024","unstructured":"Ye, S., Chen, D., Han, S., Liao, J.: 3d question answering. IEEE Trans. Visual Comput. Graphics 30(3), 1772\u20131786 (2024). https:\/\/doi.org\/10.1109\/TVCG.2022.3225327","journal-title":"IEEE Trans. Visual Comput. Graphics"},{"key":"5_CR42","unstructured":"Yin, Z., et al.: Lamm: language-assisted multi-modal instruction-tuning dataset, framework, and benchmark (2023). https:\/\/arxiv.org\/abs\/2306.06687"},{"key":"5_CR43","doi-asserted-by":"crossref","unstructured":"Yuan, Z., Ren, J., Feng, C.M., Zhao, H., Cui, S., Li, Z.: Visual programming for zero-shot open-vocabulary 3d visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 20623\u201320633 (June 2024)","DOI":"10.1109\/CVPR52733.2024.01949"},{"key":"5_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, S., et al.: Agent3d-zero: an agent for zero-shot 3d understanding (2024). https:\/\/arxiv.org\/abs\/2403.11835","DOI":"10.1007\/978-3-031-72655-2_11"},{"key":"5_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhao, Z., Zhao, Y., Wang, Q., Liu, H., Gao, L.: Where does it exist: spatio-temporal video grounding for multi-form sentences (2020). https:\/\/arxiv.org\/abs\/2001.06891","DOI":"10.1109\/CVPR42600.2020.01068"},{"key":"5_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, G., Hong, Y., Wu, Q.: Navgpt: explicit reasoning in vision-and-language navigation with large language models. arXiv preprint arXiv:2305.16986 (2023)","DOI":"10.1609\/aaai.v38i7.28597"},{"key":"5_CR47","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Ma, X., Chen, Y., Deng, Z., Huang, S., Li, Q.: 3d-vista: pre-trained transformer for 3d vision and text alignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2911\u20132921, October 2023","DOI":"10.1109\/ICCV51070.2023.00272"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91989-3_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:33:18Z","timestamp":1748194398000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91989-3_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031919886","9783031919893"],"references-count":47,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91989-3_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}