{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T04:41:22Z","timestamp":1774672882596,"version":"3.50.1"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["KJZD20230923114212024"],"award-info":[{"award-number":["KJZD20230923114212024"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["JSGG20210802152548034"],"award-info":[{"award-number":["JSGG20210802152548034"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s10489-025-06680-2","type":"journal-article","created":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T02:00:49Z","timestamp":1751248849000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["PointCoCa: Point contrastive captioner pre-trained with prompt-driven datasets generation enhances point cloud shape understanding"],"prefix":"10.1007","volume":"55","author":[{"given":"Fenglin","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianliang","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1031-5447","authenticated-orcid":false,"given":"Dong","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Xi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"6680_CR1","unstructured":"Radford A, Kim JW, Hallacy C, et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763"},{"key":"6680_CR2","unstructured":"Yu J, Wang Z, Vasudevan V et\u00a0al (2022) Coca: Contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917"},{"key":"6680_CR3","doi-asserted-by":"crossref","unstructured":"Chen X, Ma H, Wan J et\u00a0al (2017) Multi-view 3d object detection network for autonomous driving. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1907\u20131915","DOI":"10.1109\/CVPR.2017.691"},{"key":"6680_CR4","doi-asserted-by":"crossref","unstructured":"K\u00e4stner L, Frasineanu VC, Lambrecht J (2020) A 3d-deep-learning-based augmented reality calibration method for robotic environments using depth sensor data. In: 2020 IEEE international conference on robotics and automation (ICRA), IEEE, pp 1135\u20131141","DOI":"10.1109\/ICRA40945.2020.9197155"},{"key":"6680_CR5","doi-asserted-by":"crossref","unstructured":"Wang Z, Nguyen C, Asente P et\u00a0al (2023) Pointshopar: Supporting environmental design prototyping using point cloud in augmented reality. In: Proceedings of the 2023 CHI conference on human factors in computing systems, pp 1\u201315","DOI":"10.1145\/3544548.3580776"},{"key":"6680_CR6","doi-asserted-by":"crossref","unstructured":"Zhang R, Guo Z, Zhang W et\u00a0al (2022) Pointclip: Point cloud understanding by clip. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8552\u20138562","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"6680_CR7","doi-asserted-by":"crossref","unstructured":"Huang T, Dong B, Yang Y et\u00a0al (2023) Clip2point: Transfer clip to point cloud classification with image-depth pre-training. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 22157\u201322167","DOI":"10.1109\/ICCV51070.2023.02025"},{"key":"6680_CR8","unstructured":"Qi Z, Dong R, Fan G et\u00a0al (2023) Contrast with reconstruct: Contrastive 3d representation learning guided by generative pretraining. In: International conference on machine learning, PMLR, pp 28223\u201328243"},{"key":"6680_CR9","doi-asserted-by":"crossref","unstructured":"Xue L, Gao M, Xing C et\u00a0al (2023) Ulip: Learning a unified representation of language, images, and point clouds for 3d understanding. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1179\u20131189","DOI":"10.1109\/CVPR52729.2023.00120"},{"key":"6680_CR10","doi-asserted-by":"crossref","unstructured":"Jain A, Mildenhall B, Barron JT et\u00a0al (2022) Zero-shot text-guided object generation with dream fields. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 867\u2013876","DOI":"10.1109\/CVPR52688.2022.00094"},{"key":"6680_CR11","doi-asserted-by":"crossref","unstructured":"Hong F, Zhang M, Pan L et\u00a0al (2022) Avatarclip: Zero-shot text-driven generation and animation of 3d avatars. arXiv preprint arXiv:2205.08535","DOI":"10.1145\/3528223.3530094"},{"key":"6680_CR12","unstructured":"Jun H, Nichol A (2023) Shap-e: Generating conditional 3d implicit functions. arXiv preprint arXiv:2305.02463"},{"key":"6680_CR13","unstructured":"Poole B, Jain A, Barron JT et\u00a0al (2022) Dreamfusion: Text-to-3d using 2d diffusion. arXiv preprint arXiv:2209.14988"},{"key":"6680_CR14","unstructured":"Liu M, Shi R, Kuang K et\u00a0al (2024) Openshape: Scaling up 3d shape representation towards open-world understanding. Adv Neural Inf Process Syst 36"},{"key":"6680_CR15","doi-asserted-by":"crossref","unstructured":"Chen YC, Li L, Yu L et\u00a0al (2020) Uniter: Universal image-text representation learning. In: European conference on computer vision, Springer, pp 104\u2013120","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"6680_CR16","unstructured":"Jia C, Yang Y, Xia Y et\u00a0al (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, PMLR, pp 4904\u20134916"},{"key":"6680_CR17","doi-asserted-by":"crossref","unstructured":"Zhu X, Zhang R, He B et\u00a0al (2022) Pointclip V2: adapting CLIP for powerful 3d open-world learning. CoRR arXiv:2211.11682","DOI":"10.1109\/ICCV51070.2023.00249"},{"key":"6680_CR18","unstructured":"Qi CR, Su H, Mo K et\u00a0al (2017) Pointnet: Deep learning on point sets for 3d classification and segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 652\u2013660"},{"key":"6680_CR19","unstructured":"Wu Z, Song S, Khosla A et\u00a0al (2015) 3d shapenets: A deep representation for volumetric shapes. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1912\u20131920"},{"key":"6680_CR20","doi-asserted-by":"crossref","unstructured":"Maturana D, Scherer S (2015) Voxnet: A 3d convolutional neural network for real-time object recognition. In: 2015 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp 922\u2013928","DOI":"10.1109\/IROS.2015.7353481"},{"key":"6680_CR21","doi-asserted-by":"crossref","unstructured":"Su H, Maji S, Kalogerakis E et\u00a0al (2015) Multi-view convolutional neural networks for 3d shape recognition. In: Proceedings of the IEEE international conference on computer vision, pp 945\u2013953","DOI":"10.1109\/ICCV.2015.114"},{"key":"6680_CR22","unstructured":"Qi CR, Yi L, Su H et\u00a0al (2017) Pointnet++: Deep hierarchical feature learning on point sets in a metric space. Adv Neural Inf Process Syst 30"},{"key":"6680_CR23","doi-asserted-by":"crossref","unstructured":"Yu X, Tang L, Rao Y et\u00a0al (2022) Point-bert: Pre-training 3d point cloud transformers with masked point modeling. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19313\u201319322","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"6680_CR24","unstructured":"Devlin J, Chang MW, Lee K et\u00a0al (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"6680_CR25","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et\u00a0al (2020) An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"6680_CR26","unstructured":"Morton GM (1966) A computer oriented geodetic data base and a new technique in file sequencing"},{"key":"6680_CR27","unstructured":"Chen G, Wang M, Yang Y et\u00a0al (2024) Pointgpt: Auto-regressively generative pre-training from point clouds. Adv Neural Inf Process Syst 36"},{"key":"6680_CR28","doi-asserted-by":"crossref","unstructured":"Li Y, Fan H, Hu R et\u00a0al (2023) Scaling language-image pre-training via masking. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23390\u201323400","DOI":"10.1109\/CVPR52729.2023.02240"},{"key":"6680_CR29","unstructured":"Li B, Hu Y, Nie X et\u00a0al (2022) Dropkey. CoRR arXiv:2208.02646"},{"key":"6680_CR30","unstructured":"Touvron H, Martin L, Stone K et\u00a0al (2023) Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288"},{"key":"6680_CR31","unstructured":"Nichol A, Jun H, Dhariwal P et\u00a0al (2022) Point-e: A system for generating 3d point clouds from complex prompts. arXiv preprint arXiv:2212.08751"},{"key":"6680_CR32","unstructured":"Krasin I, Duerig T, Alldrin N et\u00a0al (2016) Openimages: A public dataset for large-scale multi-label and multi-class image classification. Dataset available from https:\/\/www.githubcom\/openimages"},{"key":"6680_CR33","doi-asserted-by":"crossref","unstructured":"Kuznetsova A, Rom H, Alldrin N et\u00a0al (2020) The open images dataset v4: Unified image classification, object detection, and visual relationship detection at scale. IJCV","DOI":"10.1007\/s11263-020-01316-z"},{"key":"6680_CR34","doi-asserted-by":"crossref","unstructured":"Deitke M, Schwenk D, Salvador J et\u00a0al (2023) Objaverse: A universe of annotated 3d objects. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 13142\u201313153","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"6680_CR35","doi-asserted-by":"crossref","unstructured":"Uy MA, Pham QH, Hua BS et\u00a0al (2019) Revisiting point cloud classification: A new benchmark dataset and classification model on real-world data. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1588\u20131597","DOI":"10.1109\/ICCV.2019.00167"},{"key":"6680_CR36","doi-asserted-by":"crossref","unstructured":"Hegde D, Valanarasu JMJ, Patel V (2023) Clip goes 3d: Leveraging prompt tuning for language grounded 3d recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2028\u20132038","DOI":"10.1109\/ICCVW60793.2023.00217"},{"key":"6680_CR37","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"key":"6680_CR38","unstructured":"Loshchilov I, Hutter F (2016) Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983"},{"key":"6680_CR39","doi-asserted-by":"crossref","unstructured":"Wu T, Zhang J, Fu X et\u00a0al (2023) Omniobject3d: Large-vocabulary 3d object dataset for realistic perception, reconstruction and generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 803\u2013814","DOI":"10.1109\/CVPR52729.2023.00084"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06680-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06680-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06680-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T15:57:52Z","timestamp":1758297472000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06680-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":39,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["6680"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06680-2","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"29 May 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 June 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}],"article-number":"829"}}