{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T16:50:38Z","timestamp":1783615838945,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758219","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"12784-12791","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Open3D-VQA: A Benchmark for Embodied Spatial Concept Reasoning with Multimodal Large Language Model in Open Space"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-8892-9389","authenticated-orcid":false,"given":"Weichen","family":"Zhang","sequence":"first","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6979-4761","authenticated-orcid":false,"given":"Zile","family":"Zhou","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2543-1495","authenticated-orcid":false,"given":"Xin","family":"Zeng","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9076-4819","authenticated-orcid":false,"given":"Liu","family":"Xuchen","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2599-9929","authenticated-orcid":false,"given":"Jianjie","family":"Fang","sequence":"additional","affiliation":[{"name":"Northeastern University, Qinhuangdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7561-5646","authenticated-orcid":false,"given":"Chen","family":"Gao","sequence":"additional","affiliation":[{"name":"BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7833-1876","authenticated-orcid":false,"given":"Jinqiang","family":"Cui","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5617-1659","authenticated-orcid":false,"given":"Yong","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Tsinghua University, Beijing, China and BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8271-5023","authenticated-orcid":false,"given":"Xinlei","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5241-0069","authenticated-orcid":false,"given":"Xiao-Ping","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shenzhen Key Laboratory of Ubiquitous Data Enabling, Tsinghua University, Shenzhen, China and Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00387"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01854"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00319"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01370"},{"key":"e_1_3_2_1_6_1","volume-title":"Ddl: Empowering delivery drones with large-scale urban sensing capability","author":"Chen Xuecheng","year":"2024","unstructured":"Xuecheng Chen, Haoyang Wang, Yuhan Cheng, Haohao Fu, Yuxuan Liu, Fan Dang, Yunhao Liu, Jinqiang Cui, and Xinlei Chen. 2024a. Ddl: Empowering delivery drones with large-scale urban sensing capability. IEEE Journal of Selected Topics in Signal Processing (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544793.3560412"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3389771"},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024b. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_1_10_1","volume-title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Model. arXiv preprint arXiv:2406.01584","author":"Cheng An-Chieh","year":"2024","unstructured":"An-Chieh Cheng, Hongxu Yin, Yang Fu, Qiushan Guo, Ruihan Yang, Jan Kautz, Xiaolong Wang, and Sifei Liu. 2024. SpatialRGPT: Grounded Spatial Reasoning in Vision Language Model. arXiv preprint arXiv:2406.01584 (2024)."},{"key":"e_1_3_2_1_11_1","unstructured":"Alibaba Cloud. 2025. Qwen Documentation. https:\/\/tongyi.aliyun.com\/. Accessed: 2025-01-24."},{"key":"e_1_3_2_1_12_1","volume-title":"Conference on robot learning. PMLR, 245-255","author":"Driess Danny","year":"2022","unstructured":"Danny Driess, Jung-Su Ha, Marc Toussaint, and Russ Tedrake. 2022. Learning models as functionals of signed-distance fields for manipulation planning. In Conference on robot learning. PMLR, 245-255."},{"key":"e_1_3_2_1_13_1","volume-title":"Embspatial-bench: Benchmarking spatial understanding for embodied tasks with large vision-language models. arXiv preprint arXiv:2406.05756","author":"Du Mengfei","year":"2024","unstructured":"Mengfei Du, Binhao Wu, Zejun Li, Xuanjing Huang, and Zhongyu Wei. 2024. Embspatial-bench: Benchmarking spatial understanding for embodied tasks with large vision-language models. arXiv preprint arXiv:2406.05756 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"WildUAV: Monocular UAV Dataset for Depth Estimation Tasks. 2021 IEEE 17th International Conference on Intelligent Computer Communication and Processing (ICCP)","author":"Florea Horatiu","year":"2021","unstructured":"Horatiu Florea, Vlad-Cristian Miclea, and Sergiu Nedevschi. 2021. WildUAV: Monocular UAV Dataset for Depth Estimation Tasks. 2021 IEEE 17th International Conference on Intelligent Computer Communication and Processing (ICCP) (2021)."},{"key":"e_1_3_2_1_15_1","unstructured":"Chen Gao Baining Zhao Weichen Zhang Jinzhu Mao Jun Zhang Zhiheng Zheng Fanhang Man Jianjie Fang Zile Zhou Jinqiang Cui et al. 2024. EmbodiedCity: A Benchmark Platform for Embodied Agent in Real-world City Environment. arXiv preprint arXiv:2410.09604 (2024)."},{"key":"e_1_3_2_1_16_1","unstructured":"Google. 2025. Gemini API Documentation. https:\/\/ai.google.dev\/gemini-api\/docs. Accessed: 2025-01-24."},{"key":"e_1_3_2_1_17_1","volume-title":"MR-COGraphs: Communication-efficient Multi-Robot Open-vocabulary Mapping System via 3D Scene Graphs","author":"Gu Qiuyi","year":"2025","unstructured":"Qiuyi Gu, Zhaocheng Ye, Jincheng Yu, Jiahao Tang, Tinghao Yi, Yuhan Dong, Jian Wang, Jinqiang Cui, Xinlei Chen, and Yu Wang. 2025. MR-COGraphs: Communication-efficient Multi-Robot Open-vocabulary Mapping System via 3D Scene Graphs. IEEE Robotics and Automation Letters (2025)."},{"key":"e_1_3_2_1_18_1","volume-title":"DriveMLLM: A Benchmark for Spatial Understanding with Multimodal Large Language Models in Autonomous Driving. arXiv preprint arXiv:2411.13112","author":"Guo Xianda","year":"2024","unstructured":"Xianda Guo, Ruijun Zhang, Yiqun Duan, Yuhang He, Chenming Zhang, Shuai Liu, and Long Chen. 2024. DriveMLLM: A Benchmark for Spatial Understanding with Multimodal Large Language Models in Autonomous Driving. arXiv preprint arXiv:2411.13112 (2024)."},{"key":"e_1_3_2_1_19_1","first-page":"20482","article-title":"3d-llm: Injecting the 3d world into large language models","volume":"36","author":"Hong Yining","year":"2023","unstructured":"Yining Hong, Haoyu Zhen, Peihao Chen, Shuhong Zheng, Yilun Du, Zhenfang Chen, and Chuang Gan. 2023. 3d-llm: Injecting the 3d world into large language models. Advances in Neural Information Processing Systems, Vol. 36 (2023), 20482-20494.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","volume-title":"An embodied generalist agent in 3d world. arXiv preprint arXiv:2311.12871","author":"Huang Jiangyong","year":"2023","unstructured":"Jiangyong Huang, Silong Yong, Xiaojian Ma, Xiongkun Linghu, Puhao Li, Yan Wang, Qing Li, Song-Chun Zhu, Baoxiong Jia, and Siyuan Huang. 2023. An embodied generalist agent in 3d world. arXiv preprint arXiv:2311.12871 (2023)."},{"key":"e_1_3_2_1_21_1","volume-title":"Rekep: Spatio-temporal reasoning of relational keypoint constraints for robotic manipulation. arXiv preprint arXiv:2409.01652","author":"Huang Wenlong","year":"2024","unstructured":"Wenlong Huang, Chen Wang, Yunzhu Li, Ruohan Zhang, and Li Fei-Fei. 2024. Rekep: Spatio-temporal reasoning of relational keypoint constraints for robotic manipulation. arXiv preprint arXiv:2409.01652 (2024)."},{"key":"e_1_3_2_1_22_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"ICLR 2025 Workshop on Embodied Intelligence with Large Language Models In Open City Environment.","author":"Jian Zhuozhu","unstructured":"Zhuozhu Jian, Xuran Pu, Jianjie Fang, Zhiyuan Deng, Xueqian Wang, Xinlei Chen, et al., [n.d.]. A Large Language Model-Driven Heterogeneous Air-Ground Search Swarm. In ICLR 2025 Workshop on Embodied Intelligence with Large Language Models In Open City Environment."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58604-1_7"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20074-8_6"},{"key":"e_1_3_2_1_27_1","volume-title":"Multi-modal situated reasoning in 3d scenes. arXiv preprint arXiv:2409.02389","author":"Linghu Xiongkun","year":"2024","unstructured":"Xiongkun Linghu, Jiangyong Huang, Xuesong Niu, Xiaojian Ma, Baoxiong Jia, and Siyuan Huang. 2024. Multi-modal situated reasoning in 3d scenes. arXiv preprint arXiv:2409.02389 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00566"},{"key":"e_1_3_2_1_29_1","volume-title":"Visual instruction tuning. Advances in neural information processing systems","author":"Liu Haotian","year":"2024","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2024. Visual instruction tuning. Advances in neural information processing systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01411"},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Machine Learning. PMLR, 23033-23044","author":"Luo Huaishao","year":"2023","unstructured":"Huaishao Luo, Junwei Bao, Youzheng Wu, Xiaodong He, and Tianrui Li. 2023. Segclip: Patch aggregation with learnable centers for open-vocabulary semantic segmentation. In International Conference on Machine Learning. PMLR, 23033-23044."},{"key":"e_1_3_2_1_32_1","volume-title":"Sqa3d: Situated question answering in 3d scenes. arXiv preprint arXiv:2210.07474","author":"Ma Xiaojian","year":"2022","unstructured":"Xiaojian Ma, Silong Yong, Zilong Zheng, Qing Li, Yitao Liang, Song-Chun Zhu, and Siyuan Huang. 2022. Sqa3d: Situated question answering in 3d scenes. arXiv preprint arXiv:2210.07474 (2022)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01298"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02495"},{"key":"e_1_3_2_1_35_1","unstructured":"Nikhila Ravi Valentin Gabeur Yuan-Ting Hu Ronghang Hu Chaitanya Ryali Tengyu Ma Haitham Khedr Roman R\u00e4dle Chloe Rolland Laura Gustafson et al. 2024. Sam 2: Segment anything in images and videos. arXiv preprint arXiv:2408.00714 (2024)."},{"key":"e_1_3_2_1_36_1","unstructured":"RemyXAI. 2023. VQASynth: A Framework for Synthetic Visual Question Answering Dataset Generation. https:\/\/github.com\/remyxai\/VQASynth Accessed: 2024-05-21."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3594739.3612905"},{"key":"e_1_3_2_1_38_1","volume-title":"Xin Yu, Gholamreza Haffari, and Yuan-Fang Li.","author":"Shiri Fatemeh","year":"2024","unstructured":"Fatemeh Shiri, Xiao-Yu Guo, Mona Golestan Far, Xin Yu, Gholamreza Haffari, and Yuan-Fang Li. 2024. An Empirical Analysis on Spatial Reasoning Capabilities of Large Multimodal Models. arXiv preprint arXiv:2411.06048 (2024)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01075"},{"key":"e_1_3_2_1_40_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"Superglue: A stickier benchmark for general-purpose language understanding systems. Advances in neural information processing systems","author":"Wang Alex","year":"2019","unstructured":"Alex Wang, Yada Pruksachatkun, Nikita Nangia, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel Bowman. 2019. Superglue: A stickier benchmark for general-purpose language understanding systems. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560905.3568432"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621375"},{"key":"e_1_3_2_1_44_1","volume-title":"Vggt: Visual geometry grounded transformer. arXiv preprint arXiv:2503.11651","author":"Wang Jianyuan","year":"2025","unstructured":"Jianyuan Wang, Minghao Chen, Nikita Karaev, Andrea Vedaldi, Christian Rupprecht, and David Novotny. 2025. Vggt: Visual geometry grounded transformer. arXiv preprint arXiv:2503.11651 (2025)."},{"key":"e_1_3_2_1_45_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024a. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Thinking in space: How multimodal large language models see, remember, and recall spaces. arXiv preprint arXiv:2412.14171","author":"Yang Jihan","year":"2024","unstructured":"Jihan Yang, Shusheng Yang, Anjali W Gupta, Rilyn Han, Li Fei-Fei, and Saining Xie. 2024. Thinking in space: How multimodal large language models see, remember, and recall spaces. arXiv preprint arXiv:2412.14171 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2025\/1200"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1511"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1558"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00272"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758219","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:02:39Z","timestamp":1765342959000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758219"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":50,"alternative-id":["10.1145\/3746027.3758219","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758219","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}