{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:30:30Z","timestamp":1777656630080,"version":"3.51.4"},"publisher-location":"Cham","reference-count":54,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031727634","type":"print"},{"value":"9783031727641","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72764-1_26","type":"book-chapter","created":{"date-parts":[[2024,10,24]],"date-time":"2024-10-24T14:03:10Z","timestamp":1729778590000},"page":"456-473","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Any2Point: Empowering Any-Modality Large Models for\u00a0Efficient 3D Understanding"],"prefix":"10.1007","author":[{"given":"Yiwen","family":"Tang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ray","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaming","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zoey","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhigang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuelong","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,25]]},"reference":[{"key":"26_CR1","first-page":"16664","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen, S., Ge, C., Tong, Z., Wang, J., Song, Y., Wang, J., Luo, P.: Adaptformer: adapting vision transformers for scalable visual recognition. Adv. Neural. Inf. Process. Syst. 35, 16664\u201316678 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"26_CR2","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: Scannet: Richly-annotated 3d reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839 (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"26_CR3","unstructured":"Dehghani, M., et\u00a0al.: Scaling vision transformers to 22 billion parameters. In: International Conference on Machine Learning, pp. 7480\u20137512. PMLR (2023)"},{"key":"26_CR4","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"26_CR5","unstructured":"Dong, R., et al.: Autoencoders as cross-modal teachers: Can pretrained 2d image transformers help 3d representation learning? arXiv preprint arXiv:2212.08320 (2022)"},{"key":"26_CR6","doi-asserted-by":"publisher","first-page":"681","DOI":"10.1007\/s11023-020-09548-1","volume":"30","author":"L Floridi","year":"2020","unstructured":"Floridi, L., Chiriatti, M.: Gpt-3: its nature, scope, limits, and consequences. Mind. Mach. 30, 681\u2013694 (2020)","journal-title":"Mind. Mach."},{"key":"26_CR7","doi-asserted-by":"crossref","unstructured":"Girdhar, R., et al.: Imagebind: one embedding space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15180\u201315190 (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"26_CR8","doi-asserted-by":"crossref","unstructured":"Gong, Y., Lai, C.I., Chung, Y.A., Glass, J.: Ssast: self-supervised audio spectrogram transformer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 10699\u201310709 (2022)","DOI":"10.1609\/aaai.v36i10.21315"},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"Guo, Z., Zhang, R., Qiu, L., Li, X., Heng, P.A.: Joint-mae: 2d-3d joint masked autoencoders for 3d point cloud pre-training. arXiv preprint arXiv:2302.14007 (2023)","DOI":"10.24963\/ijcai.2023\/88"},{"key":"26_CR10","doi-asserted-by":"crossref","unstructured":"Guo, Z., et al.: Viewrefer: grasp the multi-view knowledge for 3d visual grounding with gpt and prototype guidance. arXiv preprint arXiv:2303.16894 (2023)","DOI":"10.1109\/ICCV51070.2023.01410"},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"26_CR12","unstructured":"Houlsby, N., et al.: Parameter-efficient transfer learning for nlp. In: International Conference on Machine Learning, pp. 2790\u20132799. PMLR (2019)"},{"key":"26_CR13","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"26_CR14","doi-asserted-by":"crossref","unstructured":"Jia, M., et al.: Visual prompt tuning. In: European Conference on Computer Vision, pp. 709\u2013727. Springer (2022)","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"26_CR15","doi-asserted-by":"crossref","unstructured":"Jing, L., et al.: X4d-sceneformer: enhanced scene understanding on 4d point cloud videos through cross-modal knowledge transfer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 2670\u20132678 (2024)","DOI":"10.1609\/aaai.v38i3.28045"},{"key":"26_CR16","unstructured":"Li, F., et al.: Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895 (2024)"},{"key":"26_CR17","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Manipllm: embodied multimodal large language model for object-centric robotic manipulation. arXiv preprint arXiv:2312.16217 (2023)","DOI":"10.1109\/CVPR52733.2024.01710"},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: P-tuning v2: prompt tuning can be comparable to fine-tuning universally across scales and tasks. arXiv preprint arXiv:2110.07602 (2021)","DOI":"10.18653\/v1\/2022.acl-short.8"},{"key":"26_CR19","unstructured":"Liu, Y., et al.: Roberta: a robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"26_CR20","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"26_CR21","doi-asserted-by":"crossref","unstructured":"Ma, C., Liu, Y.L., Wang, Z., Liu, W., Liu, X., Wang, Z.: Humannerf-se: a simple yet effective approach to animate humannerf with diverse poses. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1460\u20131470 (2024)","DOI":"10.1109\/CVPR52733.2024.00145"},{"key":"26_CR22","unstructured":"Ma, X., Qin, C., You, H., Ran, H., Fu, Y.: Rethinking network design and local geometry in point cloud: a simple residual mlp framework. arXiv preprint arXiv:2202.07123 (2022)"},{"key":"26_CR23","unstructured":"OpenAI: GPT-4 technical report (2023)"},{"key":"26_CR24","unstructured":"Oquab, M., et\u00a0al.: Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"26_CR25","doi-asserted-by":"publisher","unstructured":"Pang, Y., Wang, W., Tay, F.E., Liu, W., Tian, Y., Yuan, L.: Masked autoencoders for point cloud self-supervised learning. In: European Conference on Computer Vision, pp. 604\u2013621. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_35","DOI":"10.1007\/978-3-031-20086-1_35"},{"key":"26_CR26","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3d classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"26_CR27","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J.: Pointnet++: Deep hierarchical feature learning on point sets in a metric space. Advances in neural information processing systems 30 (2017)"},{"key":"26_CR28","unstructured":"Qi, Z., et al.: Contrast with reconstruct: Contrastive 3d representation learning guided by generative pretraining. arXiv preprint arXiv:2302.02318 (2023)"},{"key":"26_CR29","first-page":"23192","volume":"35","author":"G Qian","year":"2022","unstructured":"Qian, G., Li, Y., Peng, H., Mai, J., Hammoud, H., Elhoseiny, M., Ghanem, B.: Pointnext: revisiting pointnet++ with improved training and scaling strategies. Adv. Neural. Inf. Process. Syst. 35, 23192\u201323204 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"26_CR30","unstructured":"Qu, D., et al.: Livescene: language embedding interactive radiance fields for physical scene rendering and control. arXiv preprint arXiv:2406.16038 (2024)"},{"key":"26_CR31","doi-asserted-by":"crossref","unstructured":"Qu, D., et al.: Implicit event-rgbd neural slam. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19584\u201319594 (2024)","DOI":"10.1109\/CVPR52733.2024.01852"},{"key":"26_CR32","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"issue":"1","key":"26_CR33","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1), 5485\u20135551 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"26_CR34","doi-asserted-by":"crossref","unstructured":"Tang, Y., et al.: Point-peft: parameter-efficient fine-tuning for 3d pre-trained models. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 5171\u20135179 (2024)","DOI":"10.1609\/aaai.v38i6.28323"},{"key":"26_CR35","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, pp. 10347\u201310357. PMLR (2021)"},{"key":"26_CR36","doi-asserted-by":"crossref","unstructured":"Uy, M.A., Pham, Q.H., Hua, B.S., Nguyen, T., Yeung, S.K.: Revisiting point cloud classification: a new benchmark dataset and classification model on real-world data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1588\u20131597 (2019)","DOI":"10.1109\/ICCV.2019.00167"},{"issue":"5","key":"26_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3326362","volume":"38","author":"Y Wang","year":"2019","unstructured":"Wang, Y., Sun, Y., Liu, Z., Sarma, S.E., Bronstein, M.M., Solomon, J.M.: Dynamic graph cnn for learning on point clouds. ACM Trans. Graph. (tog) 38(5), 1\u201312 (2019)","journal-title":"ACM Trans. Graph. (tog)"},{"key":"26_CR38","first-page":"14388","volume":"35","author":"Z Wang","year":"2022","unstructured":"Wang, Z., Yu, X., Rao, Y., Zhou, J., Lu, J.: P2p: tuning pre-trained image models for point cloud analysis with point-to-pixel prompting. Adv. Neural. Inf. Process. Syst. 35, 14388\u201314402 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"26_CR39","doi-asserted-by":"crossref","unstructured":"Wei, C., Fan, H., Xie, S., Wu, C.Y., Yuille, A., Feichtenhofer, C.: Masked feature prediction for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14668\u201314678 (2022)","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"26_CR40","unstructured":"Wu, Z., et al.: 3d shapenets: a deep representation for volumetric shapes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1912\u20131920 (2015)"},{"key":"26_CR41","doi-asserted-by":"crossref","unstructured":"Xie, Z., et al.: Simmim: a simple framework for masked image modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9653\u20139663 (2022)","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"26_CR42","doi-asserted-by":"crossref","unstructured":"Xue, L., et al.: Ulip: learning a unified representation of language, images, and point clouds for 3d understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1179\u20131189 (2023)","DOI":"10.1109\/CVPR52729.2023.00120"},{"key":"26_CR43","unstructured":"Yang, S., et al.: Lidar-llm: exploring the potential of large language models for 3d lidar understanding. arXiv preprint arXiv:2312.14074 (2023)"},{"key":"26_CR44","doi-asserted-by":"crossref","unstructured":"Yu, X., Tang, L., Rao, Y., Huang, T., Zhou, J., Lu, J.: Point-bert: pre-training 3d point cloud transformers with masked point modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19313\u201319322 (2022)","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"26_CR45","doi-asserted-by":"crossref","unstructured":"Zha, Y., Wang, J., Dai, T., Chen, B., Wang, Z., Xia, S.T.: Instance-aware dynamic prompt tuning for pre-trained point cloud models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.01302"},{"key":"26_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, D., et al.: Fm-ov3d: foundation model-based cross-modal knowledge blending for open-vocabulary 3d detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 16723\u201316731 (2024)","DOI":"10.1609\/aaai.v38i15.29612"},{"key":"26_CR47","first-page":"27061","volume":"35","author":"R Zhang","year":"2022","unstructured":"Zhang, R., Guo, Z., Gao, P., Fang, R., Zhao, B., Wang, D., Qiao, Y., Li, H.: Point-m2ae: multi-scale masked autoencoders for hierarchical point cloud pre-training. Adv. Neural. Inf. Process. Syst. 35, 27061\u201327074 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"26_CR48","doi-asserted-by":"crossref","unstructured":"Zhang, R., et al.: Pointclip: point cloud understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8552\u20138562 (2022)","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"26_CR49","unstructured":"Zhang, R., et al.: Llama-adapter: Efficient fine-tuning of language models with zero-init attention. arXiv preprint arXiv:2303.16199 (2023)"},{"key":"26_CR50","unstructured":"Zhang, R., et al.: Personalize segment anything model with one shot. arXiv preprint arXiv:2305.03048 (2023)"},{"key":"26_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, R., Wang, L., Qiao, Y., Gao, P., Li, H.: Learning 3d representations from 2d pre-trained models via image-to-point masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21769\u201321780 (2023)","DOI":"10.1109\/CVPR52729.2023.02085"},{"key":"26_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, R., Wang, L., Wang, Y., Gao, P., Li, H., Shi, J.: Starting from non-parametric networks for 3d point cloud analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5344\u20135353 (2023)","DOI":"10.1109\/CVPR52729.2023.00517"},{"key":"26_CR53","doi-asserted-by":"crossref","unstructured":"Zhu, X., et al.: No time to train: Empowering non-parametric networks for few-shot 3d scene segmentation. arXiv preprint arXiv:2404.04050 (2024)","DOI":"10.1109\/CVPR52733.2024.00368"},{"key":"26_CR54","doi-asserted-by":"crossref","unstructured":"Zhu, X., et al.: Pointclip v2: Prompting clip and gpt for powerful 3d open-world learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2639\u20132650 (2023)","DOI":"10.1109\/ICCV51070.2023.00249"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72764-1_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,24]],"date-time":"2024-10-24T14:12:40Z","timestamp":1729779160000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72764-1_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,25]]},"ISBN":["9783031727634","9783031727641"],"references-count":54,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72764-1_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,25]]},"assertion":[{"value":"25 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}