{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:35:31Z","timestamp":1778081731199,"version":"3.51.4"},"publisher-location":"Cham","reference-count":53,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915741","type":"print"},{"value":"9783031915758","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91575-8_4","type":"book-chapter","created":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:36Z","timestamp":1748195856000},"page":"53-70","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["THP3D: Text-Driven Multi-granularity 3D Human Parsing"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6523-1948","authenticated-orcid":false,"given":"Keito","family":"Suzuki","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8039-7591","authenticated-orcid":false,"given":"Bang","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4527-1114","authenticated-orcid":false,"given":"Kunyao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9112-8146","authenticated-orcid":false,"given":"Runfa","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5022-063X","authenticated-orcid":false,"given":"Truong","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"4_CR1","doi-asserted-by":"crossref","unstructured":"Abdelreheem, A., Skorokhodov, I., Ovsjanikov, M., Wonka, P.: SATR: zero-shot semantic segmentation of 3D shapes. arXiv preprint arXiv:2304.04909 (2023)","DOI":"10.1109\/ICCV51070.2023.01392"},{"key":"4_CR2","doi-asserted-by":"crossref","unstructured":"Anti\u0107, D., Tiwari, G., Ozcomlekci, B., Marin, R., Pons-Moll, G.: Close: a 3D clothing segmentation dataset and model. In: International Conference on 3D Vision (3DV) (2024)","DOI":"10.1109\/3DV62453.2024.00020"},{"key":"4_CR3","doi-asserted-by":"crossref","unstructured":"Armeni, I., et al.: 3D semantic parsing of large-scale indoor spaces. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1534\u20131543 (2016)","DOI":"10.1109\/CVPR.2016.170"},{"key":"4_CR4","doi-asserted-by":"crossref","unstructured":"Bhatnagar, B.L., Tiwari, G., Theobalt, C., Pons-Moll, G.: Multi-garment net: learning to dress 3D people from images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5420\u20135430 (2019)","DOI":"10.1109\/ICCV.2019.00552"},{"key":"4_CR5","unstructured":"Chang, A.X., et\u00a0al.: ShapeNet: an information-rich 3D model repository. arXiv preprint arXiv:1512.03012 (2015)"},{"issue":"12","key":"4_CR6","doi-asserted-by":"publisher","first-page":"5020","DOI":"10.1109\/TVCG.2022.3197383","volume":"29","author":"K Chen","year":"2023","unstructured":"Chen, K., Yin, F., Du, B., Wu, B., Nguyen, T.Q.: Efficient registration for human surfaces via isometric regularization on embedded deformation. IEEE Trans. Visual Comput. Graphics 29(12), 5020\u20135032 (2023). https:\/\/doi.org\/10.1109\/TVCG.2022.3197383","journal-title":"IEEE Trans. Visual Comput. Graphics"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Chen, R., et al.: CLIP2Scene: towards label-efficient 3D scene understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7020\u20137030 (2023)","DOI":"10.1109\/CVPR52729.2023.00678"},{"key":"4_CR8","doi-asserted-by":"crossref","unstructured":"Chen, X., Mottaghi, R., Liu, X., Fidler, S., Urtasun, R., Yuille, A.: Detect what you can: detecting and representing objects using holistic models and body parts. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1971\u20131978 (2014)","DOI":"10.1109\/CVPR.2014.254"},{"issue":"1","key":"4_CR9","first-page":"1","volume":"41","author":"X Chen","year":"2021","unstructured":"Chen, X., Pang, A., Yang, W., Wang, P., Xu, L., Yu, J.: TightCap: 3D human shape capture with clothing tightness field. ACM Trans. Graph. (TOG) 41(1), 1\u201317 (2021)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"4_CR10","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A.G., Kirillov, A., Girdhar, R.: Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1290\u20131299 (2022)","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"4_CR11","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. Adv. Neural. Inf. Process. Syst. 34, 17864\u201317875 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR12","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: ScanNet: richly-annotated 3D reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5828\u20135839 (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"4_CR13","doi-asserted-by":"crossref","unstructured":"Ding, J., Xue, N., Xia, G.S., Dai, D.: Decoupling zero-shot semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11583\u201311592 (2022)","DOI":"10.1109\/CVPR52688.2022.01129"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Dong, Q., et al.: Laplacian2mesh: Laplacian-based mesh understanding. IEEE Trans. Vis. Comput. Graph. (2023)","DOI":"10.1109\/TVCG.2023.3259044"},{"key":"4_CR15","doi-asserted-by":"crossref","unstructured":"Ghiasi, G., Gu, X., Cui, Y., Lin, T.Y.: Scaling open-vocabulary image segmentation with image-level labels. In: European Conference on Computer Vision, pp. 540\u2013557. Springer (2022)","DOI":"10.1007\/978-3-031-20059-5_31"},{"issue":"4","key":"4_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3306346.3322959","volume":"38","author":"R Hanocka","year":"2019","unstructured":"Hanocka, R., Hertz, A., Fish, N., Giryes, R., Fleishman, S., Cohen-Or, D.: MeshCNN: a network with an edge. ACM Trans. Graph. (ToG) 38(4), 1\u201312 (2019)","journal-title":"ACM Trans. Graph. (ToG)"},{"key":"4_CR17","unstructured":"He, Y., et al.: Deep learning based 3D segmentation: a survey. arXiv preprint arXiv:2103.05423 (2021)"},{"key":"4_CR18","first-page":"27940","volume":"34","author":"F Hong","year":"2021","unstructured":"Hong, F., Pan, L., Cai, Z., Liu, Z.: Garment4D: garment reconstruction from point cloud sequences. Adv. Neural. Inf. Process. Syst. 34, 27940\u201327951 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR19","doi-asserted-by":"crossref","unstructured":"Huang, T., et al.: Clip2point: transfer clip to point cloud classification with image-depth pre-training. arXiv preprint arXiv:2210.01055 (2022)","DOI":"10.1109\/ICCV51070.2023.02025"},{"key":"4_CR20","doi-asserted-by":"crossref","unstructured":"Jertec, A., Bojani\u0107, D., Bartol, K., Pribani\u0107, T., Petkovi\u0107, T., Petrak, S.: On using pointnet architecture for human body segmentation. In: 2019 11th International Symposium on Image and Signal Processing and Analysis (ISPA), pp. 253\u2013257. IEEE (2019)","DOI":"10.1109\/ISPA.2019.8868844"},{"key":"4_CR21","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"4_CR22","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"4_CR23","unstructured":"Li, B., Weinberger, K.Q., Belongie, S., Koltun, V., Ranftl, R.: Language-driven semantic segmentation. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=RriDjddCLN"},{"key":"4_CR24","doi-asserted-by":"crossref","unstructured":"Li, L.H., et\u00a0al.: Grounded language-image pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10965\u201310975 (2022)","DOI":"10.1109\/CVPR52688.2022.01069"},{"issue":"6","key":"4_CR25","doi-asserted-by":"publisher","first-page":"3260","DOI":"10.1109\/TPAMI.2020.3048039","volume":"44","author":"P Li","year":"2020","unstructured":"Li, P., Xu, Y., Wei, Y., Yang, Y.: Self-correction for human parsing. IEEE Trans. Pattern Anal. Mach. Intell. 44(6), 3260\u20133271 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4_CR26","doi-asserted-by":"crossref","unstructured":"Liang, F., et al.: Open-vocabulary semantic segmentation with mask-adapted clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7061\u20137070 (2023)","DOI":"10.1109\/CVPR52729.2023.00682"},{"issue":"4","key":"4_CR27","doi-asserted-by":"publisher","first-page":"871","DOI":"10.1109\/TPAMI.2018.2820063","volume":"41","author":"X Liang","year":"2018","unstructured":"Liang, X., Gong, K., Shen, X., Lin, L.: Look into person: joint body parsing & pose estimation network and a new benchmark. IEEE Trans. Pattern Anal. Mach. Intell. 41(4), 871\u2013885 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4_CR28","doi-asserted-by":"crossref","unstructured":"Liu, M., et al.: Partslip: low-shot part segmentation for 3D point clouds via pretrained image-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21736\u201321746 (2023)","DOI":"10.1109\/CVPR52729.2023.02082"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Loper, M., Mahmood, N., Romero, J., Pons-Moll, G., Black, M.J.: SMPL: a skinned multi-person linear model. ACM Trans. Graphics (Proc. SIGGRAPH Asia) 34(6), 248:1\u2013248:16 (2015)","DOI":"10.1145\/2816795.2818013"},{"issue":"4","key":"4_CR30","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1145\/3072959.3073616","volume":"36","author":"H Maron","year":"2017","unstructured":"Maron, H., et al.: Convolutional neural networks on surfaces via seamless toric covers. ACM Trans. Graph. 36(4), 71\u20131 (2017)","journal-title":"ACM Trans. Graph."},{"key":"4_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.gmod.2023.101187","volume":"129","author":"P Musoni","year":"2023","unstructured":"Musoni, P., Melzi, S., Castellani, U.: GIM3D plus: a labeled 3D dataset to design data-driven solutions for dressed humans. Graph. Models 129, 101187 (2023)","journal-title":"Graph. Models"},{"key":"4_CR32","unstructured":"Paszke, A., et\u00a0al.: PyTorch: an imperative style, high-performance deep learning library. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"4_CR33","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: PointNet: deep learning on point sets for 3D classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"4_CR34","unstructured":"Qi, C.R., Yi, L., Su, H., Guibas, L.J.: PointNet++: deep hierarchical feature learning on point sets in a metric space. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"4_CR35","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"4_CR36","unstructured":"Renderpeople: (2023). https:\/\/renderpeople.com\/3d-people"},{"key":"4_CR37","doi-asserted-by":"publisher","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"4_CR38","doi-asserted-by":"crossref","unstructured":"Schult, J., Engelmann, F., Hermans, A., Litany, O., Tang, S., Leibe, B.: Mask3D: mask transformer for 3D semantic instance segmentation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 8216\u20138223. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160590"},{"issue":"3","key":"4_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3507905","volume":"41","author":"N Sharp","year":"2022","unstructured":"Sharp, N., Attaiki, S., Crane, K., Ovsjanikov, M.: DiffusionNet: discretization agnostic learning on surfaces. ACM Trans. Graph. (TOG) 41(3), 1\u201316 (2022)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"4_CR40","doi-asserted-by":"crossref","unstructured":"Takmaz, A., et al.: 3D segmentation of humans in point clouds with synthetic data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1292\u20131304 (2023)","DOI":"10.1109\/ICCV51070.2023.00125"},{"key":"4_CR41","unstructured":"Tangseng, P., Wu, Z., Yamaguchi, K.: Looking at outfit to parse clothing. arXiv preprint arXiv:1703.01386 (2017)"},{"key":"4_CR42","doi-asserted-by":"crossref","unstructured":"Thomas, H., Qi, C.R., Deschaud, J.E., Marcotegui, B., Goulette, F., Guibas, L.J.: KPConv: flexible and deformable convolution for point clouds. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6411\u20136420 (2019)","DOI":"10.1109\/ICCV.2019.00651"},{"key":"4_CR43","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-030-58580-8_1","volume-title":"Computer Vision \u2013 ECCV 2020","author":"G Tiwari","year":"2020","unstructured":"Tiwari, G., Bhatnagar, B.L., Tung, T., Pons-Moll, G.: SIZER: a dataset and model for parsing 3D clothing and learning size sensitive 3D clothing. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12348, pp. 1\u201318. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58580-8_1"},{"key":"4_CR44","doi-asserted-by":"crossref","unstructured":"Ueshima, T., Hotta, K., Tokai, S., Zhang, C.: Training pointnet for human point cloud segmentation with 3d meshes. In: Fifteenth International Conference on Quality Control by Artificial Vision, vol. 11794, pp. 72\u201377. SPIE (2021)","DOI":"10.1117\/12.2589075"},{"key":"4_CR45","unstructured":"Wu, Z., et al.: 3D shapenets: a deep representation for volumetric shapes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1912\u20131920 (2015)"},{"key":"4_CR46","doi-asserted-by":"crossref","unstructured":"Xu, M., Ding, R., Zhao, H., Qi, X.: PAConv: position adaptive convolution with dynamic kernel assembling on point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3173\u20133182 (2021)","DOI":"10.1109\/CVPR46437.2021.00319"},{"key":"4_CR47","doi-asserted-by":"crossref","unstructured":"Yu, T., Zheng, Z., Guo, K., Liu, P., Dai, Q., Liu, Y.: Function4D: real-time human volumetric capture from very sparse consumer RGBD sensors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5746\u20135756 (2021)","DOI":"10.1109\/CVPR46437.2021.00569"},{"key":"4_CR48","doi-asserted-by":"crossref","unstructured":"Yuksel, C.: Sample elimination for generating poisson disk sample sets. In: Computer Graphics Forum, vol.\u00a034, pp. 25\u201332. Wiley Online Library (2015)","DOI":"10.1111\/cgf.12538"},{"key":"4_CR49","doi-asserted-by":"crossref","unstructured":"Zeng, Y., et al.: CLIP2: contrastive language-image-point pretraining from real-world point cloud data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15244\u201315253 (2023)","DOI":"10.1109\/CVPR52729.2023.01463"},{"key":"4_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, R., et al.: PointCLIP: point cloud understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8552\u20138562 (2022)","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"4_CR51","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jiang, L., Jia, J., Torr, P.H., Koltun, V.: Point transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16259\u201316268 (2021)","DOI":"10.1109\/ICCV48922.2021.01595"},{"key":"4_CR52","unstructured":"Zhou, Q., Liu, Y., Yu, C., Li, J., Wang, Z., Wang, F.: LMSeg: language-guided multi-dataset segmentation. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"4_CR53","doi-asserted-by":"crossref","unstructured":"Zhu, X., et al.: PointCLIP V2: prompting CLIP and GPT for powerful 3D open-world learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2639\u20132650 (2023)","DOI":"10.1109\/ICCV51070.2023.00249"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91575-8_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:49Z","timestamp":1748195869000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91575-8_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915741","9783031915758"],"references-count":53,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91575-8_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}