{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T18:32:01Z","timestamp":1771612321645,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758177","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"12492-12500","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Queryable 3D Scene Representation: A Multi-Modal Framework for Semantic Reasoning and Robotic Task Planning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1717-0669","authenticated-orcid":false,"given":"Xun","family":"Li","sequence":"first","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5273-7296","authenticated-orcid":false,"given":"Rodrigo","family":"Santa Cruz","sequence":"additional","affiliation":[{"name":"CSIRO, Pullenvale, QLD, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1291-4136","authenticated-orcid":false,"given":"Mingze","family":"Xi","sequence":"additional","affiliation":[{"name":"CSIRO, Canberra, ACT, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9892-9515","authenticated-orcid":false,"given":"Hu","family":"Zhang","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0928-0380","authenticated-orcid":false,"given":"Madhawa","family":"Perera","sequence":"additional","affiliation":[{"name":"CSIRO, Canberra, ACT, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0107-7347","authenticated-orcid":false,"given":"Ziwei","family":"Wang","sequence":"additional","affiliation":[{"name":"CSIRO, Pullenvale, QLD, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8022-9208","authenticated-orcid":false,"given":"Ahalya","family":"Ravendran","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8673-2434","authenticated-orcid":false,"given":"Brandon J.","family":"Matthews","sequence":"additional","affiliation":[{"name":"CSIRO, Canberra, ACT, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3932-1093","authenticated-orcid":false,"given":"Feng","family":"Xu","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6191-9887","authenticated-orcid":false,"given":"Matt","family":"Adcock","sequence":"additional","affiliation":[{"name":"CSIRO, Canberra, ACT, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0409-2259","authenticated-orcid":false,"given":"Dadong","family":"Wang","sequence":"additional","affiliation":[{"name":"CSIRO, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8160-1796","authenticated-orcid":false,"given":"Jiajun","family":"Liu","sequence":"additional","affiliation":[{"name":"CSIRO, Pullenvale, QLD, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Deep ViT features as dense visual descriptors. arXiv preprint arXiv:2112.05814","author":"Amir Shir","year":"2021","unstructured":"Shir Amir, Yossi Gandelsman, Shai Bagon, and Tali Dekel. 2021. Deep ViT features as dense visual descriptors. arXiv preprint arXiv:2112.05814, Vol. 2, 3 (2021), 4."},{"key":"e_1_3_2_1_2_1","volume-title":"Robotics: Science and Systems","author":"Bandyopadhyay T.","year":"2024","unstructured":"T. Bandyopadhyay, F. Talbot, C. Bennie, H. Senaratne, X. Li, B. Tidd, M. Xi, J. Stiefel, V. Dedeoglu, R. Taylor, T. Molnar, Z. Wang, J. Pinskier, F. Xu, L. Liow, B. Burgess-Limerick, J. Haviland, P. Sikka, S. Murrell, J. Hodgkinson, J. Liu, F. Pauling, and S. Funiak. 2024. Demonstrating Event-Triggered Investigation and Sample Collection for Human Scientists using Field Robots and Large Foundation Models. In Robotics: Science and Systems. RSS Foundation, Delft, Netherlands."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_4_1","first-page":"4247","article-title":"Object goal navigation using goal-oriented semantic exploration","volume":"33","author":"Chaplot Devendra Singh","year":"2020","unstructured":"Devendra Singh Chaplot, Dhiraj Gandhi, Abhinav Gupta, and Ruslan R Salakhutdinov. 2020. Object goal navigation using goal-oriented semantic exploration. Advances in Neural Information Processing Systems, Vol. 33 (2020), 4247-4258.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161534"},{"key":"e_1_3_2_1_6_1","volume-title":"2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE, 5802-5808","author":"Chen X.","unstructured":"X. Chen, I. Vizzo, T. L\u00e4be, J. Behley, and C. Stachniss. 2021. Range image-based LiDAR localization for autonomous vehicles. In 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE, 5802-5808."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Bowen Cheng Ishan Misra Alexander G. Schwing Alexander Kirillov and Rohit Girdhar. 2022. Masked-attention Mask Transformer for Universal Image Segmentation. CVPR.","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"e_1_3_2_1_8_1","volume-title":"Mohammadreza Salehi, Niklas Muennighoff, Kyle Lo, Luca Soldaini, et al.","author":"Deitke Matt","year":"2024","unstructured":"Matt Deitke, Christopher Clark, Sangho Lee, Rohun Tripathi, Yue Yang, Jae Sung Park, Mohammadreza Salehi, Niklas Muennighoff, Kyle Lo, Luca Soldaini, et al., 2024. Molmo and pixmo: Open weights and open data for state-of-the-art multimodal models. arXiv preprint arXiv:2409.17146 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"S. Fredriksson A. Saradagi and G. Nikolakopoulos. 2024. Voxel Map to Occupancy Map Conversion Using Free Space Projection for Efficient Map Representation for Aerial and Ground Robots. IEEE Robotics and Automation Letters (2024).","DOI":"10.1109\/LRA.2024.3495575"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610243"},{"key":"e_1_3_2_1_12_1","volume-title":"Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:2104.13921","author":"Gu Xiuye","year":"2021","unstructured":"Xiuye Gu, Tsung-Yi Lin, Weicheng Kuo, and Yin Cui. 2021. Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:2104.13921 (2021)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Karan M Jatavallabhula Aditya Kuwajerwala et al. 2023. ConceptFusion: Open-set multimodal 3d mapping. arXiv preprint arXiv:2302.07241 (2023).","DOI":"10.15607\/RSS.2023.XIX.066"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"e_1_3_2_1_16_1","unstructured":"A. Juliani. 2018. Unity: A general platform for intelligent agents. arXiv preprint arXiv:1809.02627."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01807"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01253"},{"key":"e_1_3_2_1_20_1","volume-title":"International Conference on Machine Learning. PMLR, 12888-12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven CH Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In International Conference on Machine Learning. PMLR, 12888-12900."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2015.2430526"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2017.7989538"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967890"},{"key":"e_1_3_2_1_25_1","volume-title":"International Conference on Machine Learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, et al., 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_26_1","volume-title":"Wildcat: Online continuous-time 3D lidar-inertial SLAM. arXiv preprint arXiv:2205.12595","author":"Ramezani M.","year":"2022","unstructured":"M. Ramezani, K. Khosoussi, G. Catt, P. Moghadam, J. Williams, P. Borges, and N. Kottege. 2022. Wildcat: Online continuous-time 3D lidar-inertial SLAM. arXiv preprint arXiv:2205.12595 (2022)."},{"key":"e_1_3_2_1_27_1","volume-title":"Clip-fields: Weakly supervised semantic fields for robotic memory. arXiv preprint arXiv:2210.05663","author":"Shafiullah Nawid","year":"2022","unstructured":"Nawid Shafiullah, Chris Paxton, et al., 2022. Clip-fields: Weakly supervised semantic fields for robotic memory. arXiv preprint arXiv:2210.05663 (2022)."},{"key":"e_1_3_2_1_28_1","volume-title":"Conference on Robot Learning. PMLR, 492-504","author":"Shah Dhruv","year":"2023","unstructured":"Dhruv Shah, Bartlomiej Osinski, and Sergey Levine. 2023. LM-Nav: Robotic navigation with large pre-trained models of language, vision, and action. In Conference on Robot Learning. PMLR, 492-504."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00873"},{"key":"e_1_3_2_1_30_1","unstructured":"Julian Straub Thomas Whelan Lingni Ma Yufan Chen Erik Wijmans Simon Green Jakob J Engel Raul Mur-Artal Carl Ren Shobhit Verma et al. 2019. The Replica dataset: A digital replica of indoor spaces. arXiv preprint arXiv:1906.05797 (2019)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2020.104032"},{"key":"e_1_3_2_1_32_1","volume-title":"PLGS: Robust Panoptic Lifting with 3D Gaussian Splatting. arXiv preprint arXiv:2410.17505","author":"Wang Yifan","year":"2024","unstructured":"Yifan Wang, Xun Wei, Ming Lu, and Gang Kang. 2024. PLGS: Robust Panoptic Lifting with 3D Gaussian Splatting. arXiv preprint arXiv:2410.17505 (2024)."},{"key":"e_1_3_2_1_33_1","unstructured":"Shitao Xiao Zheng Liu Peitian Zhang and Niklas Muennighoff. 2023. C-Pack: Packaged Resources To Advance General Chinese Embedding. arXiv:2309.07597 [cs.CL]"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.544"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758177","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:14:53Z","timestamp":1765307693000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758177"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":34,"alternative-id":["10.1145\/3746027.3758177","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758177","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}