{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:07:00Z","timestamp":1765339620311,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62332016"],"award-info":[{"award-number":["62332016"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002367","name":"Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["ZDBS-LY-JSC001"],"award-info":[{"award-number":["ZDBS-LY-JSC001"]}],"id":[{"id":"10.13039\/501100002367","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755735","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:55:00Z","timestamp":1761375300000},"page":"5070-5079","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PGOV3D: Open-Vocabulary 3D Semantic Segmentation with Partial-to-Global Curriculum"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6886-2529","authenticated-orcid":false,"given":"Shiqi","family":"Zhang","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1814-6783","authenticated-orcid":false,"given":"Sha","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9624-7451","authenticated-orcid":false,"given":"Jiajun","family":"Deng","sequence":"additional","affiliation":[{"name":"The University of Adelaide, Adelaide, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2762-2822","authenticated-orcid":false,"given":"Yedong","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2901-0988","authenticated-orcid":false,"given":"Mingxiao","family":"Ma","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6520-255X","authenticated-orcid":false,"given":"Yanyong","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"8748","volume-title":"PmLR","author":"Radford A.","year":"2021","unstructured":"A. Radford, J. W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark et al., ''Learning transferable visual models from natural language supervision,'' in International conference on machine learning. PmLR, 2021, pp. 8748--8763."},{"key":"e_1_3_2_1_2_1","first-page":"815","article-title":"Openscene: 3d scene understanding with open vocabularies","author":"Peng S.","year":"2023","unstructured":"S. Peng, K. Genova, C. Jiang, A. Tagliasacchi, M. Pollefeys, T. Funkhouser et al., ''Openscene: 3d scene understanding with open vocabularies,'' in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2023, pp. 815--824.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_3_1","first-page":"21","volume-title":"Angelova et al., ''3d open-vocabulary panoptic segmentation with 2d-3d vision-language distillation,'' in European Conference on Computer Vision","author":"Xiao Z.","year":"2024","unstructured":"Z. Xiao, L. Jing, S. Wu, A. Z. Zhu, J. Ji, C. M. Jiang, W.-C. Hung, T. Funkhouser, W. Kuo, A. Angelova et al., ''3d open-vocabulary panoptic segmentation with 2d-3d vision-language distillation,'' in European Conference on Computer Vision. Springer, 2024, pp. 21--38."},{"key":"e_1_3_2_1_4_1","volume-title":"Ov3d: Open-vocabulary 3d object detection,'' in IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Yang J.","year":"2023","unstructured":"J. Yang, X. Li, Y. Liu, J. Gu, W. Zhang, and C. Liu, ''Ov3d: Open-vocabulary 3d object detection,'' in IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"Language-driven open-vocabulary 3d scene understanding","author":"Ding R.","year":"2022","unstructured":"R. Ding, J. Yang, C. Xue, W. Zhang, S. Bai, and X. Qi, ''Language-driven open-vocabulary 3d scene understanding,'' Nov 2022."},{"key":"e_1_3_2_1_6_1","first-page":"296","article-title":"Improved baselines with visual instruction tuning","author":"Liu H.","year":"2024","unstructured":"H. Liu, C. Li, Y. Li, and Y. J. Lee, ''Improved baselines with visual instruction tuning,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 26 296--26 306.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_7_1","volume-title":"Llava-next: Improved reasoning, ocr, and world knowledge","author":"Liu H.","year":"2024","unstructured":"H. Liu, C. Li, Y. Li, B. Li, Y. Zhang, S. Shen, and Y. J. Lee, ''Llava-next: Improved reasoning, ocr, and world knowledge,'' January 2024. [Online]. Available: https:\/\/llava-vl.github.io\/blog\/2024-01--30-llava-next\/"},{"key":"e_1_3_2_1_8_1","first-page":"4015","article-title":"Segment anything","author":"Kirillov A.","year":"2023","unstructured":"A. Kirillov, E. Mintun, N. Ravi, H. Mao, C. Rolland, L. Gustafson, T. Xiao, S. White-head, A. C. Berg, W.-Y. Lo et al., ''Segment anything,'' in Proceedings of the IEEE\/CVF international conference on computer vision, 2023, pp. 4015--4026.","journal-title":"Proceedings of the IEEE\/CVF international conference on computer vision"},{"key":"e_1_3_2_1_9_1","first-page":"19","volume-title":"Segment everything everywhere all at once,'' Advances in neural information processing systems","author":"Zou X.","year":"2023","unstructured":"X. Zou, J. Yang, H. Zhang, F. Li, L. Li, J. Wang, L. Wang, J. Gao, and Y. J. Lee, ''Segment everything everywhere all at once,'' Advances in neural information processing systems, vol. 36, pp. 19 769--19 782, 2023."},{"key":"e_1_3_2_1_10_1","first-page":"729","article-title":"Lerf: Language embedded radiance fields","author":"Kerr J.","year":"2023","unstructured":"J. Kerr, C. M. Kim, K. Goldberg, A. Kanazawa, and M. Tancik, ''Lerf: Language embedded radiance fields,'' in Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 19 729--19 739.","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"e_1_3_2_1_11_1","first-page":"051","article-title":"Langsplat: 3d language gaussian splatting","author":"Qin M.","year":"2024","unstructured":"M. Qin, W. Li, J. Zhou, H. Wang, and H. Pfister, ''Langsplat: 3d language gaussian splatting,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 20 051--20 060.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_12_1","first-page":"57","volume-title":"Diff3detr: Agent-based diffusion model for semi-supervised 3d object detection,'' in European Conference on Computer Vision","author":"Deng J.","year":"2024","unstructured":"J. Deng, J. Lu, and T. Zhang, ''Diff3detr: Agent-based diffusion model for semi-supervised 3d object detection,'' in European Conference on Computer Vision. Springer, 2024, pp. 57--73."},{"key":"e_1_3_2_1_13_1","first-page":"943","article-title":"Oneformer3d: One transformer for unified point cloud segmentation","author":"Kolodiazhnyi M.","year":"2024","unstructured":"M. Kolodiazhnyi, A. Vorontsova, A. Konushin, and D. Rukhovich, ''Oneformer3d: One transformer for unified point cloud segmentation,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 20 943--20 953.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_14_1","first-page":"4490","article-title":"Voxelnet: End-to-end learning for point cloud based 3d object detection","author":"Zhou Y.","year":"2018","unstructured":"Y. Zhou and O. Tuzel, ''Voxelnet: End-to-end learning for point cloud based 3d object detection,'' in Proceedings of the IEEE conference on computer vision and pattern recognition, 2018, pp. 4490--4499.","journal-title":"Proceedings of the IEEE conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_15_1","first-page":"4840","article-title":"Point transformer v3: Simpler faster stronger","author":"Wu X.","year":"2024","unstructured":"X. Wu, L. Jiang, P.-S. Wang, Z. Liu, X. Liu, Y. Qiao, W. Ouyang, T. He, and H. Zhao, ''Point transformer v3: Simpler faster stronger,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 4840--4851.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_16_1","first-page":"259","article-title":"Point transformer","author":"Zhao H.","year":"2021","unstructured":"H. Zhao, L. Jiang, J. Jia, P. H. Torr, and V. Koltun, ''Point transformer,'' in Proceedings of the IEEE\/CVF international conference on computer vision, 2021, pp. 16 259--16 268.","journal-title":"Proceedings of the IEEE\/CVF international conference on computer vision"},{"key":"e_1_3_2_1_17_1","first-page":"4867","article-title":"Pointgroup: Dual-set point grouping for 3d instance segmentation","author":"Jiang L.","year":"2020","unstructured":"L. Jiang, H. Zhao, S. Shi, S. Liu, C.-W. Fu, and J. Jia, ''Pointgroup: Dual-set point grouping for 3d instance segmentation,'' in Proceedings of the IEEE\/CVF conference on computer vision and Pattern recognition, 2020, pp. 4867--4876.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and Pattern recognition"},{"key":"e_1_3_2_1_18_1","first-page":"3693","article-title":"Mask-attention-free transformer for 3d instance segmentation","author":"Lai X.","year":"2023","unstructured":"X. Lai, Y. Yuan, R. Chu, Y. Chen, H. Hu, and J. Jia, ''Mask-attention-free transformer for 3d instance segmentation,'' in Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 3693--3703.","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"e_1_3_2_1_19_1","first-page":"516","article-title":"Query refinement transformer for 3d instance segmentation","author":"Lu J.","year":"2023","unstructured":"J. Lu, J. Deng, C. Wang, J. He, and T. Zhang, ''Query refinement transformer for 3d instance segmentation,'' in Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 18 516--18 526.","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25335"},{"key":"e_1_3_2_1_21_1","first-page":"2708","article-title":"Softgroup for 3d instance segmentation on point clouds","author":"Vu T.","year":"2022","unstructured":"T. Vu, K. Kim, T. M. Luu, T. Nguyen, and C. D. Yoo, ''Softgroup for 3d instance segmentation on point clouds,'' in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2022, pp. 2708--2717.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_22_1","first-page":"975","article-title":"Ins-conv: Incremental sparse convolution for online 3d segmentation","author":"Liu L.","year":"2022","unstructured":"L. Liu, T. Zheng, Y.-J. Lin, K. Ni, and L. Fang, ''Ins-conv: Incremental sparse convolution for online 3d segmentation,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 18 975--18 984.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_23_1","first-page":"4534","article-title":"point convolution for online semantic 3d scene segmentation","author":"Zhang J.","year":"2020","unstructured":"J. Zhang, C. Zhu, L. Zheng, and K. Xu, ''Fusion-aware point convolution for online semantic 3d scene segmentation,'' in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2020, pp. 4534--4543.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_24_1","first-page":"5575","article-title":"Learning multi-view aggregation in the wild for large-scale 3d semantic segmentation","author":"Robert D.","year":"2022","unstructured":"D. Robert, B. Vallet, and L. Landrieu, ''Learning multi-view aggregation in the wild for large-scale 3d semantic segmentation,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 5575--5584.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_25_1","first-page":"8216","volume-title":"IEEE","author":"Schult J.","year":"2023","unstructured":"J. Schult, F. Engelmann, A. Hermans, O. Litany, S. Tang, and B. Leibe, ''Mask3d: Mask transformer for 3d semantic instance segmentation,'' in 2023 IEEE International Conference on Robotics and Automation (ICRA). IEEE, 2023, pp. 8216--8223."},{"key":"e_1_3_2_1_26_1","first-page":"7020","volume-title":"Clip2scene: Towards label-efficient 3d scene understanding by clip","author":"Chen R.","year":"2023","unstructured":"R. Chen, Y. Liu, L. Kong, X. Zhu, Y. Ma, Y. Li, Y. Hou, Y. Qiao, and W. Wang, ''Clip2scene: Towards label-efficient 3d scene understanding by clip,'' pp. 7020--7030, 2023."},{"key":"e_1_3_2_1_27_1","first-page":"901","article-title":"Yolo-world: Real-time open-vocabulary object detection","author":"Cheng T.","year":"2024","unstructured":"T. Cheng, L. Song, Y. Ge, W. Liu, X. Wang, and Y. Shan, ''Yolo-world: Real-time open-vocabulary object detection,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 16 901--16 911.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_28_1","first-page":"7010","article-title":"Language-driven open-vocabulary 3d scene understanding","author":"Ding R.","year":"2023","unstructured":"R. Ding, J. Yang, C. Xue, W. Zhang, S. Bai, and X. Qi, ''Pla: Language-driven open-vocabulary 3d scene understanding,'' in IEEE Conference on Computer Vision and Pattern Recognition, 2023, pp. 7010--7019.","journal-title":"IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_29_1","volume-title":"Language-driven semantic segmentation,'' arXiv preprint arXiv:2201.03546","author":"Li B.","year":"2022","unstructured":"B. Li, K. Q. Weinberger, S. Belongie, V. Koltun, and R. Ranftl, ''Language-driven semantic segmentation,'' arXiv preprint arXiv:2201.03546, 2022."},{"key":"e_1_3_2_1_30_1","first-page":"7061","article-title":"Open-vocabulary semantic segmentation with mask-adapted clip","author":"Liang F.","year":"2023","unstructured":"F. Liang, B. Wu, X. Dai, K. Li, Y. Zhao, H. Zhang, P. Zhang, P. Vajda, and D. Marculescu, ''Open-vocabulary semantic segmentation with mask-adapted clip,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recogni- tion, 2023, pp. 7061--7070.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recogni- tion"},{"key":"e_1_3_2_1_31_1","first-page":"815","article-title":"Openseene: 3d scene understanding with open vocabularies","author":"Peng S.","year":"2023","unstructured":"S. Peng, K. Genova, C. Jiang, A. Tagliasacchi, M. Pollefeys, T. Funkhouser et al., ''Openseene: 3d scene understanding with open vocabularies,'' in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2023, pp. 815--824.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_32_1","first-page":"540","article-title":"open-vocabulary image segmentation with image-level labels","author":"Ghiasi G.","year":"2022","unstructured":"G. Ghiasi, X. Gu, Y. Cui, and T.-Y. Lin, ''Scaling open-vocabulary image segmentation with image-level labels,'' European Conference on Computer Vision, pp. 540--557, 2022.","journal-title":"European Conference on Computer Vision"},{"key":"e_1_3_2_1_33_1","volume-title":"Semantic gaussians: Open-vocabulary scene understanding with 3d gaussian splatting,'' arXiv preprint arXiv:2403.15624","author":"Guo J.","year":"2024","unstructured":"J. Guo, X. Ma, Y. Fan, H. Liu, and Q. Li, ''Semantic gaussians: Open-vocabulary scene understanding with 3d gaussian splatting,'' arXiv preprint arXiv:2403.15624, 2024."},{"key":"e_1_3_2_1_34_1","first-page":"2945","article-title":"Side adapter network for open-vocabulary semantic segmentation","author":"Xu M.","year":"2023","unstructured":"M. Xu, Z. Zhang, F. Wei, H. Hu, and X. Bai, ''Side adapter network for open-vocabulary semantic segmentation,'' Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2945--2954, 2023.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_35_1","volume-title":"Openins3d: Snap and lookup for 3d open-vocabulary instance segmentation,'' in Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Meng Q.","year":"2023","unstructured":"Q. Meng, J. Yang, X. Li, Y. Liu, J. Gu, W. Zhang, and C. Liu, ''Openins3d: Snap and lookup for 3d open-vocabulary instance segmentation,'' in Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023."},{"key":"e_1_3_2_1_36_1","first-page":"442","article-title":"Open vocabulary 3d scene understanding via geometry guided self-distillation","author":"Wang P.","year":"2024","unstructured":"P. Wang, Y. Wang, S. Li, Z. Zhang, Z. Lei, and L. Zhang, ''Open vocabulary 3d scene understanding via geometry guided self-distillation,'' in European Conference on Computer Vision, 2024, pp. 442--460.","journal-title":"European Conference on Computer Vision"},{"key":"e_1_3_2_1_37_1","first-page":"357","article-title":"Open-vocabulary 3d semantic segmentation with text-to-image diffusion models","author":"Zhu X.","year":"2024","unstructured":"X. Zhu, H. Zhou, P. Xing, L. Zhao, H. Xu, J. Liang, A. Hauptmann, T. Liu, and A. Gallagher, ''Open-vocabulary 3d semantic segmentation with text-to-image diffusion models,'' in European Conference on Computer Vision, 2024, pp. 357--375.","journal-title":"European Conference on Computer Vision"},{"key":"e_1_3_2_1_38_1","first-page":"162","article-title":"Segment and edit anything in 3d scenes","author":"Ye M.","year":"2024","unstructured":"M. Ye, M. Danelljan, F. Yu, and L. Ke, ''Gaussian grouping: Segment and edit anything in 3d scenes,'' European Conference on Computer Vision, pp. 162--179, 2024.","journal-title":"European Conference on Computer Vision"},{"key":"e_1_3_2_1_39_1","volume-title":"Og- gaussian: Occupancy based street gaussians for autonomous driving,'' arXiv preprint arXiv:2502.14235","author":"Shen Y.","year":"2025","unstructured":"Y. Shen, X. Zhang, Y. Duan, S. Zhang, H. Li, Y. Wu, J. Ji, and Y. Zhang, ''Og- gaussian: Occupancy based street gaussians for autonomous driving,'' arXiv preprint arXiv:2502.14235, 2025."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"e_1_3_2_1_42_1","first-page":"24","volume-title":"Zhou et al., ''Chain-of-thought prompting elicits reasoning in large language models,'' Advances in neural information processing systems","author":"Wei J.","year":"2022","unstructured":"J. Wei, X. Wang, D. Schuurmans, M. Bosma, F. Xia, E. Chi, Q. V. Le, D. Zhou et al., ''Chain-of-thought prompting elicits reasoning in large language models,'' Advances in neural information processing systems, vol. 35, pp. 24 824--24 837, 2022."},{"key":"e_1_3_2_1_43_1","volume-title":"Grounded sam: Assembling open-world models for diverse visual tasks","author":"Ren T.","year":"2024","unstructured":"T. Ren, S. Liu, A. Zeng, J. Lin, K. Li, H. Cao, J. Chen, X. Huang, Y. Chen, F. Yan, Z. Zeng, H. Zhang, F. Li, J. Yang, H. Li, Q. Jiang, and L. Zhang, ''Grounded sam: Assembling open-world models for diverse visual tasks,'' 2024."},{"key":"e_1_3_2_1_44_1","volume-title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes,'' in Proc. Computer Vision and Pattern Recognition (CVPR)","author":"Dai A.","year":"2017","unstructured":"A. Dai, A. X. Chang, M. Savva, M. Halber, T. Funkhouser, and M. Nie\u00dfner, ''Scannet: Richly-annotated 3d reconstructions of indoor scenes,'' in Proc. Computer Vision and Pattern Recognition (CVPR), IEEE, 2017."},{"key":"e_1_3_2_1_45_1","volume-title":"Language-grounded indoor 3d semantic segmentation in the wild,'' in Proceedings of the European Conference on Computer Vision (ECCV)","author":"Rozenberszki D.","year":"2022","unstructured":"D. Rozenberszki, O. Litany, and A. Dai, ''Language-grounded indoor 3d semantic segmentation in the wild,'' in Proceedings of the European Conference on Computer Vision (ECCV), 2022."},{"key":"e_1_3_2_1_46_1","first-page":"1534","volume-title":"3d semantic parsing of large-scale indoor spaces,'' in 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Armeni I.","year":"2016","unstructured":"I. Armeni, O. Sener, A. R. Zamir, H. Jiang, I. Brilakis, M. Fischer, and S. Savarese, ''3d semantic parsing of large-scale indoor spaces,'' in 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2016, pp. 1534--1543."},{"key":"e_1_3_2_1_47_1","first-page":"2879","article-title":"Mseg: A composite dataset for multi-domain semantic segmentation","author":"Lambert J.","year":"2020","unstructured":"J. Lambert, Z. Liu, O. Sener, J. Hays, and V. Koltun, ''Mseg: A composite dataset for multi-domain semantic segmentation,'' in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2020, pp. 2879--2888.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"e_1_3_2_1_48_1","first-page":"823","article-title":"Regional point-language contrastive learning for open-world 3d scene understanding","author":"Yang J.","year":"2024","unstructured":"J. Yang, R. Ding, W. Deng, Z. Wang, and X. Qi, ''Regionplc: Regional point-language contrastive learning for open-world 3d scene understanding,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 19 823--19 832.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_49_1","first-page":"284","article-title":"Open-vocabulary 3d semantic segmentation with foundation models","author":"Jiang L.","year":"2024","unstructured":"L. Jiang, S. Shi, and B. Schiele, ''Open-vocabulary 3d semantic segmentation with foundation models,'' in Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 21 284--21 294.","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2019.00319"},{"key":"e_1_3_2_1_51_1","volume-title":"Nov","author":"Loshchilov I.","year":"2017","unstructured":"I. Loshchilov and F. Hutter, ''Decoupled weight decay regularization,'' Learning,Learning, Nov 2017."},{"key":"e_1_3_2_1_52_1","first-page":"992","volume-title":"IEEE","author":"Michele B.","year":"2021","unstructured":"B. Michele, A. Boulch, G. Puy, M. Bucher, and R. Marlet, ''Generative zero-shot learning for semantic segmentation of 3d point clouds,'' in 2021 International Conference on 3D Vision (3DV). IEEE, 2021, pp. 992--1002."},{"key":"e_1_3_2_1_53_1","first-page":"923","article-title":"Transductive zero-shot learning for 3d point cloud classification","author":"Cheraghian A.","year":"2020","unstructured":"A. Cheraghian, S. Rahman, D. Campbell, and L. Petersson, ''Transductive zero-shot learning for 3d point cloud classification,'' in Proceedings of the IEEE\/CVF winter conference on applications of computer vision, 2020, pp. 923--933.","journal-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2019.04.019"},{"key":"e_1_3_2_1_55_1","first-page":"922","article-title":"Cross-modal mask reasoning for open vocabulary 3d semantic segmentation","volume":"37","author":"Wang Z.","year":"2024","unstructured":"Z. Wang, Y. Wang, X. Yu, J. Zhou, and J. Lu, ''Xmask3d: Cross-modal mask reasoning for open vocabulary 3d semantic segmentation,'' Advances in Neural Information Processing Systems, vol. 37, pp. 74 922--74 944, 2024.","journal-title":"Advances in Neural Information Processing Systems"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755735","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:03:21Z","timestamp":1765339401000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755735"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":55,"alternative-id":["10.1145\/3746027.3755735","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755735","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}