{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T10:16:30Z","timestamp":1784542590196,"version":"3.55.0"},"publisher-location":"Cham","reference-count":122,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732348","type":"print"},{"value":"9783031732355","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73235-5_12","type":"book-chapter","created":{"date-parts":[[2024,9,29]],"date-time":"2024-09-29T06:01:53Z","timestamp":1727589713000},"page":"206-228","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":119,"title":["Sapiens: Foundation for\u00a0Human Vision Models"],"prefix":"10.1007","author":[{"given":"Rawal","family":"Khirodkar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Timur","family":"Bagautdinov","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Julieta","family":"Martinez","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Su","family":"Zhaoen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Austin","family":"James","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peter","family":"Selednik","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stuart","family":"Anderson","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shunsuke","family":"Saito","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,30]]},"reference":[{"key":"12_CR1","unstructured":"Abnar, S., Dehghani, M., Neyshabur, B., Sedghi, H.: Exploring the limits of large scale pre-training. arXiv preprint arXiv:2110.02095 (2021)"},{"key":"12_CR2","unstructured":"Achiam, J., et\u00a0al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Alldieck, T., Zanfir, M., Sminchisescu, C.: Photorealistic monocular 3d reconstruction of humans wearing clothing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1506\u20131515 (2022)","DOI":"10.1109\/CVPR52688.2022.00156"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Bai, Y., et al.: Sequential modeling enables scalable learning for large vision models. arXiv preprint arXiv:2312.00785 (2023)","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"12_CR5","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: Beit: bert pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021)"},{"key":"12_CR6","doi-asserted-by":"crossref","unstructured":"Barron, J.T., Malik, J.: Intrinsic scene properties from a single rgb-d image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 17\u201324 (2013)","DOI":"10.1109\/CVPR.2013.10"},{"issue":"8","key":"12_CR7","doi-asserted-by":"publisher","first-page":"1670","DOI":"10.1109\/TPAMI.2014.2377712","volume":"37","author":"JT Barron","year":"2014","unstructured":"Barron, J.T., Malik, J.: Shape, illumination, and reflectance from shading. IEEE Trans. Pattern Anal. Mach. Intell. 37(8), 1670\u20131687 (2014)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"12_CR8","doi-asserted-by":"publisher","first-page":"67281","DOI":"10.1109\/ACCESS.2021.3076595","volume":"9","author":"K Bartol","year":"2021","unstructured":"Bartol, K., Bojani\u0107, D., Petkovi\u0107, T., Pribani\u0107, T.: A review of body measurement using 3d scanning. IEEE Access 9, 67281\u201367301 (2021)","journal-title":"IEEE Access"},{"key":"12_CR9","unstructured":"Bhat, S.F., Alhashim, I., Wonka, P.: Adabins: depth estimation using adaptive bins. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4009\u20134018 (2021)"},{"key":"12_CR10","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., M\u00fcller, M.: Zoedepth: zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)"},{"key":"12_CR11","unstructured":"Birkl, R., Wofk, D., M\u00fcller, M.: Midas v3. 1\u2013a model zoo for robust monocular relative depth estimation. arXiv preprint arXiv:2307.14460 (2023)"},{"issue":"1","key":"12_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40309-022-00208-4","volume":"10","author":"L Bojic","year":"2022","unstructured":"Bojic, L.: Metaverse through the prism of power and addiction: what will happen when the virtual world becomes more attractive than reality? Eur. J. Futures Res. 10(1), 1\u201324 (2022)","journal-title":"Eur. J. Futures Res."},{"key":"12_CR13","unstructured":"Brown, T., et al.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. arXiv preprint arXiv:1611.08050 (2016)","DOI":"10.1109\/CVPR.2017.143"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.E., Sheikh, Y.: Openpose: realtime multi-person 2d pose estimation using part affinity fields. arXiv preprint arXiv:1812.08008 (2018)","DOI":"10.1109\/CVPR.2017.143"},{"key":"12_CR16","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Chan, C., Ginosar, S., Zhou, T., Efros, A.A.: Everybody dance now. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5933\u20135942 (2019)","DOI":"10.1109\/ICCV.2019.00603"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Chen, L.C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H.: Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 801\u2013818 (2018)","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"12_CR19","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"12_CR20","unstructured":"Chen, X., et\u00a0al.: Pali: a jointly-scaled multilingual language-image model. arXiv preprint arXiv:2209.06794 (2022)"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7103\u20137112 (2018)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A., Kirillov, A., Girdhar, R.: Masked-attention mask transformer for universal image segmentation. arXiv preprint arXiv:2112.01527 (2021)","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"12_CR23","unstructured":"Couprie, C., Farabet, C., Najman, L., LeCun, Y.: Indoor semantic segmentation using depth information. arXiv preprint arXiv:1301.3572 (2013)"},{"key":"12_CR24","first-page":"3965","volume":"34","author":"Z Dai","year":"2021","unstructured":"Dai, Z., Liu, H., Le, Q.V., Tan, M.: Coatnet: marrying convolution and attention for all data sizes. Adv. Neural. Inf. Process. Syst. 34, 3965\u20133977 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"12_CR25","unstructured":"Dehghani, M., et\u00a0al.: Scaling vision transformers to 22 billion parameters. In: International Conference on Machine Learning, pp. 7480\u20137512. PMLR (2023)"},{"key":"12_CR26","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"12_CR27","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Drobyshev, N., Chelishev, J., Khakhulin, T., Ivakhnenko, A., Lempitsky, V., Zakharov, E.: Megaportraits: one-shot megapixel neural head avatars. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 2663\u20132671 (2022)","DOI":"10.1145\/3503161.3547838"},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Du, Y., Kips, R., Pumarola, A., Starke, S., Thabet, A., Sanakoyeu, A.: Avatars grow legs: generating smooth human motion from sparse tracking inputs with diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 481\u2013490 (2023)","DOI":"10.1109\/CVPR52729.2023.00054"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Eftekhar, A., Sax, A., Malik, J., Zamir, A.: Omnidata: a scalable pipeline for making multi-task mid-level vision datasets from 3d scans. In: Proceedings of the IEEE\/CVF International Conference on Computer Visionm, pp. 10786\u201310796 (2021)","DOI":"10.1109\/ICCV48922.2021.01061"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Eigen, D., Fergus, R.: Predicting depth, surface normals and semantic labels with a common multi-scale convolutional architecture. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2650\u20132658 (2015)","DOI":"10.1109\/ICCV.2015.304"},{"key":"12_CR32","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Adv. Neural Inf. Process. Syst. 27 (2014)"},{"key":"12_CR33","unstructured":"El-Nouby, A., Izacard, G., Touvron, H., Laptev, I., Jegou, H., Grave, E.: Are large-scale datasets necessary for self-supervised pre-training? arXiv preprint arXiv:2112.10740 (2021)"},{"key":"12_CR34","unstructured":"El-Nouby, A., et al.: Scalable pre-training of large autoregressive image models. arXiv preprint arXiv:2401.08541 (2024)"},{"key":"12_CR35","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Lu, G., Fang, X., Xie, J., Tai, Y.W., Lu, C.: Weakly and semi supervised human body part parsing via pose-guided knowledge transfer. arXiv preprint arXiv:1805.04310 (2018)","DOI":"10.1109\/CVPR.2018.00015"},{"key":"12_CR36","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Xie, S., Tai, Y.W., Lu, C.: RMPE: regional multi-person pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2334\u20132343 (2017)","DOI":"10.1109\/ICCV.2017.256"},{"key":"12_CR37","doi-asserted-by":"crossref","unstructured":"Fang, Y., Sun, Q., Wang, X., Huang, T., Wang, X., Cao, Y.: Eva-02: a visual representation for neon genesis. arXiv preprint arXiv:2303.11331 (2023)","DOI":"10.2139\/ssrn.4813567"},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"Fang, Y., et al.: Eva: exploring the limits of masked visual representation learning at scale. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19358\u201319369 (2023)","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"12_CR39","doi-asserted-by":"crossref","unstructured":"Fouhey, D.F., Gupta, A., Hebert, M.: Data-driven 3d primitives for single image understanding. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3392\u20133399 (2013)","DOI":"10.1109\/ICCV.2013.421"},{"key":"12_CR40","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Li, Y., Chen, Y., Yang, M., Lin, L.: Instance-level human parsing via part grouping network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 770\u2013785 (2018)","DOI":"10.1007\/978-3-030-01225-0_47"},{"key":"12_CR41","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Zhang, D., Shen, X., Lin, L.: Look into person: self-supervised structure-sensitive learning and a new benchmark for human parsing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 932\u2013940 (2017)","DOI":"10.1109\/CVPR.2017.715"},{"key":"12_CR42","unstructured":"Goyal, P., et\u00a0al.: Self-supervised pretraining of visual features in the wild. arXiv preprint arXiv:2103.01988 (2021)"},{"key":"12_CR43","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Vasiljevic, I., Chen, D., Ambrus, R., Gaidon, A.: Towards zero-shot scale-aware monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9233\u20139243 (2023)","DOI":"10.1109\/ICCV51070.2023.00847"},{"key":"12_CR44","doi-asserted-by":"crossref","unstructured":"Halstead, M.A., Barsky, B.A., Klein, S.A., Mandell, R.B.: Reconstructing curved surfaces from specular reflection patterns using spline surface fitting of normals. In: Proceedings of the 23rd Annual Conference on Computer Graphics and Interactive Techniques, pp. 335\u2013342 (1996)","DOI":"10.1145\/237170.237272"},{"key":"12_CR45","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"12_CR46","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9729\u20139738 (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"12_CR47","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.B.: Mask r-cnn. arXiv preprint arXiv:1703.06870 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"12_CR48","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"12_CR49","unstructured":"Hoffmann, J., et\u00a0al.: Training compute-optimal large language models. arXiv preprint arXiv:2203.15556 (2022)"},{"key":"12_CR50","unstructured":"Hu, L., Gao, X., Zhang, P., Sun, K., Zhang, B., Bo, L.: Animate anyone: consistent and controllable image-to-video synthesis for character animation. arXiv preprint arXiv:2311.17117 (2023)"},{"key":"12_CR51","doi-asserted-by":"crossref","unstructured":"Huang, S., Gong, M., Tao, D.: A coarse-fine network for keypoint localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3028\u20133037 (2017)","DOI":"10.1109\/ICCV.2017.329"},{"key":"12_CR52","doi-asserted-by":"crossref","unstructured":"Jafarian, Y., Park, H.S.: Learning high fidelity depths of dressed humans by watching social media dance videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12753\u201312762 (2021)","DOI":"10.1109\/CVPR46437.2021.01256"},{"key":"12_CR53","unstructured":"Jiang, A.Q., et\u00a0al.: Mistral 7b. arXiv preprint arXiv:2310.06825 (2023)"},{"key":"12_CR54","unstructured":"Jiang, T., et al.: Rtmpose: real-time multi-person pose estimation based on mmpose. arXiv preprint arXiv:2303.07399 (2023)"},{"key":"12_CR55","doi-asserted-by":"publisher","unstructured":"Jin, S., et al.: Whole-body human pose estimation in the wild. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12354, pp. 196\u2013214. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_12","DOI":"10.1007\/978-3-030-58545-7_12"},{"key":"12_CR56","unstructured":"Kato, N., Li, T., Nishino, K., Uchida, Y.: Improving multi-person pose estimation using label correction. arXiv preprint arXiv:1811.03331 (2018)"},{"key":"12_CR57","doi-asserted-by":"crossref","unstructured":"Khirodkar, R., Chari, V., Agrawal, A., Tyagi, A.: Multi-instance pose networks: rethinking top-down pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3122\u20133131 (2021)","DOI":"10.1109\/ICCV48922.2021.00311"},{"key":"12_CR58","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"12_CR59","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Chang, J.H.R., Gabriel, J., Tuzel, O., Ranjan, A.: Hugs: human Gaussian splats. arXiv preprint arXiv:2311.17910 (2023)","DOI":"10.1109\/CVPR52733.2024.00055"},{"key":"12_CR60","doi-asserted-by":"crossref","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vision 123, 32\u201373 (2017)","DOI":"10.1007\/s11263-016-0981-7"},{"key":"12_CR61","unstructured":"Krizhevsky, A., Hinton, G., et\u00a0al.: Learning multiple layers of features from tiny images. arXiv preprint (2009)"},{"key":"12_CR62","doi-asserted-by":"crossref","unstructured":"Ladick\u1ef3, L., Zeisl, B., Pollefeys, M.: Discriminatively trained dense surface normal estimation. In: ECCV 2014, Part V 13, pp. 468\u2013484. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_31"},{"key":"12_CR63","doi-asserted-by":"crossref","unstructured":"Lawrence, J., et\u00a0al.: Project starline: a high-fidelity telepresence system. arXiv preprint (2021)","DOI":"10.1145\/3478513.3480490"},{"key":"12_CR64","doi-asserted-by":"crossref","unstructured":"Levoy, M., et\u00a0al.: The digital michelangelo project: 3d scanning of large statues. In: Proceedings of the 27th Annual Conference on Computer Graphics and Interactive Techniques, pp. 131\u2013144 (2000)","DOI":"10.1145\/344779.344849"},{"key":"12_CR65","unstructured":"Lewkowycz, A.: How to decay your learning rate. arXiv preprint arXiv:2103.12682 (2021)"},{"key":"12_CR66","unstructured":"Li, Z., Wang, X., Liu, X., Jiang, J.: Binsformer: revisiting adaptive bins for monocular depth estimation. arXiv preprint arXiv:2204.00987 (2022)"},{"key":"12_CR67","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"12_CR68","doi-asserted-by":"crossref","unstructured":"Liu, Z., et\u00a0al.: Swin transformer v2: scaling up capacity and resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12009\u201312019 (2022)","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"12_CR69","doi-asserted-by":"crossref","unstructured":"Lombardi, S., Simon, T., Saragih, J., Schwartz, G., Lehrmann, A., Sheikh, Y.: Neural volumes: learning dynamic renderable volumes from images. arXiv preprint arXiv:1906.07751 (2019)","DOI":"10.1145\/3306346.3323020"},{"issue":"4","key":"12_CR70","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3450626.3459863","volume":"40","author":"S Lombardi","year":"2021","unstructured":"Lombardi, S., Simon, T., Schwartz, G., Zollhoefer, M., Sheikh, Y., Saragih, J.: Mixture of volumetric primitives for efficient neural rendering. ACM Trans. Graph. 40(4), 1\u201313 (2021)","journal-title":"ACM Trans. Graph."},{"key":"12_CR71","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T., Berkeley, U.: Fully convolutional networks for semantic segmentation. arXiv preprint arXiv:1411.4038 (2014)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"12_CR72","unstructured":"Loshchilov, I., Hutter, F.: SGDR: stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)"},{"key":"12_CR73","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"issue":"3","key":"12_CR74","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1016\/0004-3702(87)90070-1","volume":"31","author":"DG Lowe","year":"1987","unstructured":"Lowe, D.G.: Three-dimensional object recognition from single two-dimensional images. Artif. Intell. 31(3), 355\u2013395 (1987)","journal-title":"Artif. Intell."},{"key":"12_CR75","doi-asserted-by":"crossref","unstructured":"Luo, Y., Zheng, Z., Zheng, L., Guan, T., Yu, J., Yang, Y.: Macro-micro adversarial network for human parsing. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 418\u2013434 (2018)","DOI":"10.1007\/978-3-030-01240-3_26"},{"key":"12_CR76","doi-asserted-by":"crossref","unstructured":"Ma, S., et al.:: Pixel codec avatars. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 64\u201373 (2021)","DOI":"10.1109\/CVPR46437.2021.00013"},{"key":"12_CR77","doi-asserted-by":"crossref","unstructured":"Mahajan, D., et al.: Exploring the limits of weakly supervised pretraining. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 181\u2013196 (2018)","DOI":"10.1007\/978-3-030-01216-8_12"},{"key":"12_CR78","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: European Conference on Computer Vision, pp. 483\u2013499. Springer (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"12_CR79","unstructured":"Oquab, M., et\u00a0al.: Dinov2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"12_CR80","doi-asserted-by":"crossref","unstructured":"Papandreou, G., et al.: Towards accurate multi-person pose estimation in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4903\u20134911 (2017)","DOI":"10.1109\/CVPR.2017.395"},{"key":"12_CR81","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"12_CR82","unstructured":"Render people: 3d people for architectural visualization | renderpeople, https:\/\/renderpeople.com\/3d-people. Accessed 22 Feb 2024"},{"key":"12_CR83","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"12_CR84","doi-asserted-by":"crossref","unstructured":"Ranftl, R., Lasinger, K., Hafner, D., Schindler, K., Koltun, V.: Towards robust monocular depth estimation: mixing datasets for zero-shot cross-dataset transfer. IEEE Trans. Pattern Anal. Mach. Intell. 44(3), 1623\u20131637 (2020)","DOI":"10.1109\/TPAMI.2020.3019967"},{"key":"12_CR85","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"12_CR86","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., et al.: Imagenet large scale visual recognition challenge. Int. J. Comput. Vision 115, 211\u2013252 (2015)","DOI":"10.1007\/s11263-015-0816-y"},{"key":"12_CR87","unstructured":"Saharia, C., et al.: Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural. Inf. Process. Syst. 35, 36479\u201336494 (2022)"},{"key":"12_CR88","doi-asserted-by":"crossref","unstructured":"Saito, S., Huang, Z., Natsume, R., Morishima, S., Kanazawa, A., Li, H.: Pifu: pixel-aligned implicit function for high-resolution clothed human digitization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2304\u20132314 (2019)","DOI":"10.1109\/ICCV.2019.00239"},{"key":"12_CR89","doi-asserted-by":"crossref","unstructured":"Saito, S., Simon, T., Saragih, J., Joo, H.: Pifuhd: multi-level pixel-aligned implicit function for high-resolution 3d human digitization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 84\u201393 (2020)","DOI":"10.1109\/CVPR42600.2020.00016"},{"key":"12_CR90","unstructured":"Schuhmann, C., et al.: Laion-5b: an open large-scale dataset for training next generation image-text models. Adv. Neural. Inf. Process. Syst. 35, 25278\u201325294 (2022)"},{"key":"12_CR91","doi-asserted-by":"crossref","unstructured":"Singh, M., et\u00a0al.: The effectiveness of MAE pre-pretraining for billion-scale pretraining. arXiv preprint arXiv:2303.13496 (2023)","DOI":"10.1109\/ICCV51070.2023.00505"},{"key":"12_CR92","doi-asserted-by":"crossref","unstructured":"Sun, C., Shrivastava, A., Singh, S., Gupta, A.: Revisiting unreasonable effectiveness of data in deep learning era. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 843\u2013852 (2017)","DOI":"10.1109\/ICCV.2017.97"},{"key":"12_CR93","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"12_CR94","unstructured":"Taori, R., et al.: Alpaca: a strong, replicable instruction-following model. Stanford Center for Research on Foundation Models 3(6), 7 (2023). https:\/\/crfmstanford.edu\/2023\/03\/13\/alpaca.html"},{"key":"12_CR95","unstructured":"Tay, Y., et al.: Scale efficiently: insights from pre-training and fine-tuning transformers. arXiv preprint arXiv:2109.10686 (2021)"},{"key":"12_CR96","unstructured":"Team, G., et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"12_CR97","doi-asserted-by":"crossref","unstructured":"Thomee, B., et al.: Yfcc100m: the new data in multimedia research. Commun. ACM 59(2), 64\u201373 (2016)","DOI":"10.1145\/2812802"},{"key":"12_CR98","doi-asserted-by":"crossref","unstructured":"Toshev, A., Szegedy, C.: Deeppose: human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131660 (2014)","DOI":"10.1109\/CVPR.2014.214"},{"key":"12_CR99","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"12_CR100","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"12_CR101","doi-asserted-by":"crossref","unstructured":"Wang, X., Fouhey, D., Gupta, A.: Designing deep networks for surface normal estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 539\u2013547 (2015)","DOI":"10.1109\/CVPR.2015.7298652"},{"key":"12_CR102","doi-asserted-by":"crossref","unstructured":"Weng, C.Y., Curless, B., Srinivasan, P.P., Barron, J.T., Kemelmacher-Shlizerman, I.: Humannerf: free-viewpoint rendering of moving people from monocular video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16210\u201316220 (2022)","DOI":"10.1109\/CVPR52688.2022.01573"},{"key":"12_CR103","unstructured":"Wu, Y., Kirillov, A., Massa, F., Lo, W.Y., Girshick, R.: Detectron2 (2019). https:\/\/github.com\/facebookresearch\/detectron2"},{"key":"12_CR104","doi-asserted-by":"crossref","unstructured":"Xia, F., Wang, P., Chen, L.C., Yuille, A.L.: Zoom better to see clearer: human and object parsing with hierarchical auto-zoom net. In: ECCV 2016, Part V 14. pp. 648\u2013663. Springer (2016)","DOI":"10.1007\/978-3-319-46454-1_39"},{"key":"12_CR105","doi-asserted-by":"crossref","unstructured":"Xia, F., Wang, P., Chen, X., Yuille, A.L.: Joint multi-person pose estimation and semantic part segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6769\u20136778 (2017)","DOI":"10.1109\/CVPR.2017.644"},{"key":"12_CR106","doi-asserted-by":"crossref","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 466\u2013481 (2018)","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"12_CR107","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: Segformer: simple and efficient design for semantic segmentation with transformers. Adv. Neural. Inf. Process. Syst. 34, 12077\u201312090 (2021)"},{"key":"12_CR108","doi-asserted-by":"crossref","unstructured":"Xiu, Y., Yang, J., Cao, X., Tzionas, D., Black, M.J.: Econ: explicit clothed humans optimized via normal integration. arXiv preprint arXiv:2212.07422 (2022)","DOI":"10.1109\/CVPR52729.2023.00057"},{"key":"12_CR109","doi-asserted-by":"crossref","unstructured":"Xiu, Y., Yang, J., Tzionas, D., Black, M.J.: Icon: implicit clothed humans obtained from normals. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13286\u201313296. IEEE (2022)","DOI":"10.1109\/CVPR52688.2022.01294"},{"key":"12_CR110","doi-asserted-by":"crossref","unstructured":"Xu, L., et al.: Zoomnas: searching for whole-body human pose estimation in the wild. IEEE Trans. Pattern Anal. Mach. Intell. 45(4), 5296\u20135313 (2022)","DOI":"10.1109\/TPAMI.2022.3197352"},{"key":"12_CR111","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose: simple vision transformer baselines for human pose estimation. Adv. Neural. Inf. Process. Syst. 35, 38571\u201338584 (2022)"},{"key":"12_CR112","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose+: vision transformer foundation model for generic body pose estimation. arXiv preprint arXiv:2212.04246 (2022)"},{"key":"12_CR113","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Xu, X., Feng, J., Zhao, H.: Depth anything: unleashing the power of large-scale unlabeled data. arXiv preprint arXiv:2401.10891 (2024)","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"12_CR114","doi-asserted-by":"crossref","unstructured":"Yang, Z., Zeng, A., Yuan, C., Li, Y.: Effective whole-body pose estimation with two-stages distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4210\u20134220 (2023)","DOI":"10.1109\/ICCVW60793.2023.00455"},{"key":"12_CR115","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R.R., Le, Q.V.: Xlnet: generalized autoregressive pretraining for language understanding. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"12_CR116","doi-asserted-by":"crossref","unstructured":"Yin, Y., Guo, C., Kaufmann, M., Zarate, J.J., Song, J., Hilliges, O.: Hi4d: 4d instance segmentation of close human interaction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17016\u201317027 (2023)","DOI":"10.1109\/CVPR52729.2023.01632"},{"key":"12_CR117","doi-asserted-by":"crossref","unstructured":"Yu, T., Zheng, Z., Guo, K., Liu, P., Dai, Q., Liu, Y.: Function4d: real-time human volumetric capture from very sparse consumer RGBD sensors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5746\u20135756 (2021)","DOI":"10.1109\/CVPR46437.2021.00569"},{"key":"12_CR118","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"12_CR119","doi-asserted-by":"crossref","unstructured":"Zhang, S.H., et al.: Pose2seg: detection free human instance segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 889\u2013898 (2019)","DOI":"10.1109\/CVPR.2019.00098"},{"key":"12_CR120","unstructured":"Zhang, X., et al.: Gpt-4v (ision) as a generalist evaluator for vision-language tasks. arXiv preprint arXiv:2311.01361 (2023)"},{"key":"12_CR121","unstructured":"Zhou, J., et al.: ibot: image bert pre-training with online tokenizer. arXiv preprint arXiv:2111.07832 (2021)"},{"key":"12_CR122","doi-asserted-by":"crossref","unstructured":"Zlateski, A., Jaroensri, R., Sharma, P., Durand, F.: On the importance of label quality for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1479\u20131487 (2018)","DOI":"10.1109\/CVPR.2018.00160"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73235-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T21:16:45Z","timestamp":1732828605000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73235-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,30]]},"ISBN":["9783031732348","9783031732355"],"references-count":122,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73235-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,30]]},"assertion":[{"value":"30 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}