{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T16:47:53Z","timestamp":1775494073253,"version":"3.50.1"},"publisher-location":"Cham","reference-count":77,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733369","type":"print"},{"value":"9783031733376","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73337-6_5","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:02:27Z","timestamp":1730329347000},"page":"74-92","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Uni3DL: A Unified Model for\u00a03D Vision-Language Understanding"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9946-7000","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7188-5884","authenticated-orcid":false,"given":"Jian","family":"Ding","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2731-2858","authenticated-orcid":false,"given":"Zhaoyang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9659-1551","authenticated-orcid":false,"given":"Mohamed","family":"Elhoseiny","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"5_CR1","first-page":"23716","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural. Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Armeni, I., et al.: 3D semantic parsing of large-scale indoor spaces. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1534\u20131543 (2016)","DOI":"10.1109\/CVPR.2016.170"},{"issue":"3","key":"5_CR3","doi-asserted-by":"publisher","first-page":"483","DOI":"10.1090\/S0273-0979-10-01294-2","volume":"47","author":"J Baez","year":"2010","unstructured":"Baez, J., Huerta, J.: The algebra of grand unified theories. Bull. Am. Math. Soc. 47(3), 483\u2013552 (2010)","journal-title":"Bull. Am. Math. Soc."},{"key":"5_CR4","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR5","doi-asserted-by":"crossref","unstructured":"Cai, D., Zhao, L., Zhang, J., Sheng, L., Xu, D.: 3djcg: a unified framework for joint dense captioning and visual grounding on 3D point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16464\u201316473 (2022)","DOI":"10.1109\/CVPR52688.2022.01597"},{"key":"5_CR6","doi-asserted-by":"publisher","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European conference on computer vision. pp. 213\u2013229. Springer (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"5_CR7","doi-asserted-by":"publisher","unstructured":"Chen, D.Z., Chang, A.X., Nie\u00dfner, M.: Scanrefer: 3D object localization in RGB-D scans using natural language. In: European Conference On Computer Vision, pp. 202\u2013221. Springer (2020). https:\/\/doi.org\/10.1007\/978-3-030-58565-5_13","DOI":"10.1007\/978-3-030-58565-5_13"},{"key":"5_CR8","unstructured":"Chen, J., Zhu, D., Shen, X., Li, X., Liu, Z., Zhang, P., Krishnamoorthi, R., Chandra, V., Xiong, Y., Elhoseiny, M.: Minigpt-v2: large language model as a unified interface for vision-language multi-task learning. arXiv preprint arXiv:2310.09478 (2023)"},{"key":"5_CR9","doi-asserted-by":"publisher","unstructured":"Chen, K., Choy, C.B., Savva, M., Chang, A.X., Funkhouser, T., Savarese, S.: Text2shape: generating shapes from natural language by learning joint embeddings. In: Computer Vision\u2013ACCV 2018: 14th Asian Conference on Computer Vision, Perth, Australia, December 2\u20136, 2018, Revised Selected Papers, Part III 14, pp. 100\u2013116. Springer (2019). https:\/\/doi.org\/10.1007\/978-3-030-20893-6_7","DOI":"10.1007\/978-3-030-20893-6_7"},{"key":"5_CR10","doi-asserted-by":"publisher","unstructured":"Chen, S., Fang, J., Zhang, Q., Liu, W., Wang, X.: Hierarchical aggregation for 3D instance segmentation. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 15447\u201315456 (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.01518","DOI":"10.1109\/ICCV48922.2021.01518"},{"key":"5_CR11","first-page":"31333","volume":"35","author":"T Chen","year":"2022","unstructured":"Chen, T., Saxena, S., Li, L., Lin, T.Y., Fleet, D.J., Hinton, G.E.: A unified sequence interface for vision tasks. Adv. Neural. Inf. Process. Syst. 35, 31333\u201331346 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Z., Hu, R., Chen, X., Nie\u00dfner, M., Chang, A.X.: Unit3D: a unified transformer for 3D dense captioning and visual grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 18109\u201318119 (2023)","DOI":"10.1109\/ICCV51070.2023.01660"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A.G., Kirillov, A., Girdhar, R.: Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1290\u20131299 (2022)","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"5_CR14","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. Adv. Neural. Inf. Process. Syst. 34, 17864\u201317875 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR15","unstructured":"Chiang, W.L.,et\u00a0al.: Vicuna: an open-source chatbot impressing GPT-4 with 90%* chatgpt quality. See https:\/\/vicuna. lmsys. Accessed 14 April 2023 (2023)"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Choy, C., Gwak, J., Savarese, S.: 4D spatio-temporal convnets: minkowski convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3075\u20133084 (2019)","DOI":"10.1109\/CVPR.2019.00319"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: Scannet: richly-annotated 3d reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839 (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"5_CR18","unstructured":"Dai, W., et al.: Instructblip: towards general-purpose vision-language models with instruction tuning (2023)"},{"key":"5_CR19","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"5_CR20","unstructured":"Guo, Z., et\u00a0al.: Point-bind & point-LLM: aligning point cloud with multi-modality for 3D understanding, generation, and instruction following. arXiv preprint arXiv:2309.00615 (2023)"},{"key":"5_CR21","doi-asserted-by":"crossref","unstructured":"Han, L., Zheng, T., Xu, L., Fang, L.: Occuseg: occupancy-aware 3D instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2940\u20132949 (2020)","DOI":"10.1109\/CVPR42600.2020.00301"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Han, Z., Shang, M., Wang, X., Liu, Y.S., Zwicker, M.: Y2seq2seq: cross-modal representation learning for 3D shape and text by joint reconstruction and prediction of view and word sequences. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a033, pp. 126\u2013133 (2019)","DOI":"10.1609\/aaai.v33i01.3301126"},{"key":"5_CR23","unstructured":"Hong, Y., Zhen, H., Chen, P., Zheng, S., Du, Y., Chen, Z., Gan, C.: 3d-llm: Injecting the 3d world into large language models. arXiv preprint arXiv:2307.12981 (2023)"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Hou, J., Dai, A., Nie\u00dfner, M.: 3d-sis: 3D semantic instance segmentation of RGB-D scans. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4421\u20134430 (2019)","DOI":"10.1109\/CVPR.2019.00455"},{"key":"5_CR25","doi-asserted-by":"crossref","unstructured":"Huang, P.H., Lee, H.H., Chen, H.T., Liu, T.L.: Text-guided graph neural networks for referring 3D instance segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a035, pp. 1610\u20131618 (2021)","DOI":"10.1609\/aaai.v35i2.16253"},{"key":"5_CR26","doi-asserted-by":"crossref","unstructured":"Huang, T., et al.: Clip2point: transfer clip to point cloud classification with image-depth pre-training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 22157\u201322167 (2023)","DOI":"10.1109\/ICCV51070.2023.02025"},{"key":"5_CR27","doi-asserted-by":"publisher","unstructured":"Ilharco, G., et al.: Openclip (2021). https:\/\/doi.org\/10.5281\/zenodo.5143773. if you use this software, please cite it as below","DOI":"10.5281\/zenodo.5143773"},{"key":"5_CR28","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference On Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Jiang, L., Zhao, H., Shi, S., Liu, S., Fu, C.W., Jia, J.: Pointgroup: dual-set point grouping for 3D instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4867\u20134876 (2020)","DOI":"10.1109\/CVPR42600.2020.00492"},{"key":"5_CR30","doi-asserted-by":"crossref","unstructured":"Lai, X., Liu, J., Jiang, L., Wang, L., Zhao, H., Liu, S., Qi, X., Jia, J.: Stratified transformer for 3d point cloud segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8500\u20138509 (2022)","DOI":"10.1109\/CVPR52688.2022.00831"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Lai, X., Yuan, Y., Chu, R., Chen, Y., Hu, H., Jia, J.: Mask-attention-free transformer for 3D instance segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3693\u20133703 (2023)","DOI":"10.1109\/ICCV51070.2023.00342"},{"issue":"4","key":"5_CR32","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1016\/0370-1573(81)90059-4","volume":"72","author":"P Langacker","year":"1981","unstructured":"Langacker, P.: Grand unified theories and proton decay. Phys. Rep. 72(4), 185\u2013385 (1981)","journal-title":"Phys. Rep."},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Li, C., Gan, Z., Yang, Z., Yang, J., Li, L., Wang, L., Gao, J.: Multimodal foundation models: From specialists to general-purpose assistants. arXiv preprint arXiv:2309.100201 (2023)","DOI":"10.1561\/9781638283379"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Li, H., et\u00a0al.: Uni-perceiver v2: a generalist model for large-scale vision and vision-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2691\u20132700 (2023)","DOI":"10.1109\/CVPR52729.2023.00264"},{"key":"5_CR35","doi-asserted-by":"crossref","unstructured":"Liang, Z., Li, Z., Xu, S., Tan, M., Jia, K.: Instance segmentation in 3D scenes using semantic superpoint tree networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2783\u20132792 (2021)","DOI":"10.1109\/ICCV48922.2021.00278"},{"key":"5_CR36","unstructured":"Lin, C.Y.: Rouge: A package for automatic evaluation of summaries. In: Text summarization branches out. pp. 74\u201381 (2004)"},{"key":"5_CR37","unstructured":"Liu, S.H., Yu, S.Y., Wu, S.C., Chen, H.T., Liu, T.L.: Learning gaussian instance segmentation in point clouds. arXiv preprint arXiv:2007.09860 (2020)"},{"key":"5_CR38","first-page":"11525","volume":"33","author":"F Locatello","year":"2020","unstructured":"Locatello, F., et al.: Object-centric learning with slot attention. Adv. Neural. Inf. Process. Syst. 33, 11525\u201311538 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR39","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"5_CR40","unstructured":"Lu, J., Clark, C., Zellers, R., Mottaghi, R., Kembhavi, A.: Unified-IO: A unified model for vision, language, and multi-modal tasks. arXiv preprint arXiv:2206.08916 (2022)"},{"key":"5_CR41","unstructured":"Luo, T., Rockwell, C., Lee, H., Johnson, J.: Scalable 3d captioning with pretrained models. arXiv preprint arXiv:2306.07279 (2023)"},{"key":"5_CR42","doi-asserted-by":"crossref","unstructured":"Misra, I., Girdhar, R., Joulin, A.: An end-to-end transformer model for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2906\u20132917 (2021)","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"5_CR43","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of The Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"5_CR44","doi-asserted-by":"crossref","unstructured":"Park, C., Jeong, Y., Cho, M., Park, J.: Fast point transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16949\u201316958 (2022)","DOI":"10.1109\/CVPR52688.2022.01644"},{"key":"5_CR45","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: Glove: Global vectors for word representation. In: Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP). pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"5_CR46","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3D classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)"},{"key":"5_CR47","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J.: Pointnet++: deep hierarchical feature learning on point sets in a metric space. Adv. Neural Inf. Proce. Syst. 30 (2017)"},{"key":"5_CR48","first-page":"23192","volume":"35","author":"G Qian","year":"2022","unstructured":"Qian, G., et al.: Pointnext: revisiting pointnet++ with improved training and scaling strategies. Adv. Neural. Inf. Process. Syst. 35, 23192\u201323204 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR49","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision (2021)"},{"key":"5_CR50","unstructured":"Radford, A., et\u00a0al.: Improving language understanding by generative pre-training (2018)"},{"key":"5_CR51","doi-asserted-by":"crossref","unstructured":"Schult, J., Engelmann, F., Hermans, A., Litany, O., Tang, S., Leibe, B.: Mask3d: mask transformer for 3D semantic instance segmentation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 8216\u20138223. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160590"},{"key":"5_CR52","doi-asserted-by":"crossref","unstructured":"Sun, J., Qing, C., Tan, J., Xu, X.: Superpoint transformer for 3D scene instance segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a037, pp. 2393\u20132401 (2023)","DOI":"10.1609\/aaai.v37i2.25335"},{"key":"5_CR53","doi-asserted-by":"crossref","unstructured":"Tang, C., Yang, X., Wu, B., Han, Z., Chang, Y.: Parts2words: Learning joint embedding of point clouds and texts by bidirectional matching between parts and words. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6884\u20136893 (2023)","DOI":"10.1109\/CVPR52729.2023.00665"},{"key":"5_CR54","doi-asserted-by":"crossref","unstructured":"Vu, T., Kim, K., Luu, T.M., Nguyen, T., Yoo, C.D.: Softgroup for 3D instance segmentation on point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2708\u20132717 (2022)","DOI":"10.1109\/CVPR52688.2022.00273"},{"key":"5_CR55","first-page":"29975","volume":"35","author":"H Wang","year":"2022","unstructured":"Wang, H., et al.: Cagroup3D: class-aware grouping for 3d object detection on point clouds. Adv. Neural. Inf. Process. Syst. 35, 29975\u201329988 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR56","unstructured":"Wang, J., et al.: Git: A generative image-to-text transformer for vision and language. arXiv preprint arXiv:2205.14100 (2022)"},{"key":"5_CR57","unstructured":"Wang, P., et al.: OFA: unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: International Conference on Machine Learning, pp. 23318\u201323340. PMLR (2022)"},{"key":"5_CR58","unstructured":"Wang, W., et\u00a0al.: Visionllm: Large language model is also an open-ended decoder for vision-centric tasks. arXiv preprint arXiv:2305.11175 (2023)"},{"key":"5_CR59","first-page":"33330","volume":"35","author":"X Wu","year":"2022","unstructured":"Wu, X., Lao, Y., Jiang, L., Liu, X., Zhao, H.: Point transformer v2: grouped vector attention and partition-based pooling. Adv. Neural. Inf. Process. Syst. 35, 33330\u201333342 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5_CR60","unstructured":"Wu, Z., et al.: 3D shapenets: a deep representation for volumetric shapes. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1912\u20131920 (2015)"},{"key":"5_CR61","doi-asserted-by":"crossref","unstructured":"Xie, Q., et al.: Venet: voting enhancement network for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3712\u20133721 (2021)","DOI":"10.1109\/ICCV48922.2021.00369"},{"key":"5_CR62","doi-asserted-by":"crossref","unstructured":"Xu, R., Wang, X., Wang, T., Chen, Y., Pang, J., Lin, D.: Pointllm: Empowering large language models to understand point clouds. arXiv preprint arXiv:2308.16911 (2023)","DOI":"10.1007\/978-3-031-72698-9_8"},{"key":"5_CR63","doi-asserted-by":"crossref","unstructured":"Xue, L., et.: Ulip: learning a unified representation of language, images, and point clouds for 3d understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1179\u20131189 (2023)","DOI":"10.1109\/CVPR52729.2023.00120"},{"key":"5_CR64","doi-asserted-by":"crossref","unstructured":"Xue, L., et al.: Ulip-2: Towards scalable multimodal pre-training for 3d understanding. arXiv preprint arXiv:2305.08275 (2023)","DOI":"10.1109\/CVPR52733.2024.02558"},{"key":"5_CR65","unstructured":"Yang, B., Wang, J., Clark, R., Hu, Q., Wang, S., Markham, A., Trigoni, N.: Learning object bounding boxes for 3d instance segmentation on point clouds. Adv. Neural Inf. Proce. Syst. 32 (2019)"},{"key":"5_CR66","unstructured":"Yang, Y.Q., et al.: Swin3d: A pretrained transformer backbone for 3d indoor scene understanding. arXiv preprint arXiv:2304.06906 (2023)"},{"key":"5_CR67","doi-asserted-by":"crossref","unstructured":"Yang, Z., Jiang, L., Sun, Y., Schiele, B., Jia, J.: A unified query-based paradigm for point cloud understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8541\u20138551 (2022)","DOI":"10.1109\/CVPR52688.2022.00835"},{"key":"5_CR68","doi-asserted-by":"publisher","unstructured":"Yang, Z., et al.: Unitab: unifying text and box outputs for grounded vision-language modeling. In: European Conference on Computer Vision, pp. 521\u2013539. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_30","DOI":"10.1007\/978-3-031-20059-5_30"},{"key":"5_CR69","doi-asserted-by":"crossref","unstructured":"Yi, L., Zhao, W., Wang, H., Sung, M., Guibas, L.J.: GSPN: generative shape proposal network for 3d instance segmentation in point cloud. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3947\u20133956 (2019)","DOI":"10.1109\/CVPR.2019.00407"},{"key":"5_CR70","unstructured":"Yuan, L., et\u00a0al.: Florence: A new foundation model for computer vision. arXiv preprint arXiv:2111.11432 (2021)"},{"key":"5_CR71","doi-asserted-by":"crossref","unstructured":"Zhang, R., et al.: Pointclip: point cloud understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8552\u20138562 (2022)","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"5_CR72","doi-asserted-by":"crossref","unstructured":"Zhao, L., Cai, D., Sheng, L., Xu, D.: 3dvg-transformer: relation modeling for visual grounding on point clouds. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2928\u20132937 (2021)","DOI":"10.1109\/ICCV48922.2021.00292"},{"key":"5_CR73","doi-asserted-by":"publisher","unstructured":"Zheng, J., Zhang, J., Li, J., Tang, R., Gao, S., Zhou, Z.: Structured3d: a large photo-realistic dataset for structured 3D modeling. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IX 16, pp. 519\u2013535. Springer (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_30","DOI":"10.1007\/978-3-030-58545-7_30"},{"key":"5_CR74","doi-asserted-by":"crossref","unstructured":"Zhong, M., Chen, X., Chen, X., Zeng, G., Wang, Y.: Maskgroup: hierarchical point grouping and masking for 3D instance segmentation. In: 2022 IEEE International Conference on Multimedia and Expo (ICME), pp.\u00a01\u20136. IEEE (2022)","DOI":"10.1109\/ICME52920.2022.9859996"},{"key":"5_CR75","doi-asserted-by":"crossref","unstructured":"Zhu, X., et al.: Pointclip v2: prompting clip and GPT for powerful 3D open-world learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2639\u20132650 (2023)","DOI":"10.1109\/ICCV51070.2023.00249"},{"key":"5_CR76","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Ma, X., Chen, Y., Deng, Z., Huang, S., Li, Q.: 3D-vista: pre-trained transformer for 3D vision and text alignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2911\u20132921 (2023)","DOI":"10.1109\/ICCV51070.2023.00272"},{"key":"5_CR77","doi-asserted-by":"crossref","unstructured":"Zou, X., et al.: Generalized decoding for pixel, image, and language. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15116\u201315127 (2023)","DOI":"10.1109\/CVPR52729.2023.01451"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73337-6_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:03:30Z","timestamp":1730329410000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73337-6_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733369","9783031733376"],"references-count":77,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73337-6_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}