{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T16:17:47Z","timestamp":1780762667158,"version":"3.54.1"},"publisher-location":"Cham","reference-count":56,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030012489","type":"print"},{"value":"9783030012496","type":"electronic"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-030-01249-6_46","type":"book-chapter","created":{"date-parts":[[2018,10,5]],"date-time":"2018-10-05T15:35:46Z","timestamp":1538753746000},"page":"765-782","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":124,"title":["Unsupervised Geometry-Aware Representation for 3D Human Pose Estimation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2692-0801","authenticated-orcid":false,"given":"Helge","family":"Rhodin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8347-8637","authenticated-orcid":false,"given":"Mathieu","family":"Salzmann","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6702-9970","authenticated-orcid":false,"given":"Pascal","family":"Fua","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2018,10,6]]},"reference":[{"key":"46_CR1","doi-asserted-by":"crossref","unstructured":"Bas, A., Huber, P., Smith, W., Awais, M., Kittler, J.: 3D morphable models as spatial transformer networks. arXiv Preprint (2017)","DOI":"10.1109\/ICCVW.2017.110"},{"key":"46_CR2","doi-asserted-by":"crossref","unstructured":"Chen, W., et al.: Synthesizing training images for boosting human 3D pose estimation. In: 3DV (2016)","DOI":"10.1109\/3DV.2016.58"},{"key":"46_CR3","unstructured":"Chen, X., Duan, Y., Houthooft, R., Schulman, J., Sutskever, I., Abbeel, P.: Infogan: interpretable representation learning by information maximizing generative adversarial nets. In: Advances in Neural Information Processing Systems, pp. 2172\u20132180 (2016)"},{"key":"46_CR4","unstructured":"Cohen, T., Welling, M.: Transformation properties of learned visual representations. arXiv Preprint (2014)"},{"key":"46_CR5","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., Springenberg, J., Brox, T.: Learning to generate chairs with convolutional neural networks. In: Conference on Computer Vision and Pattern Recognition (2015)","DOI":"10.1109\/CVPR.2015.7298761"},{"issue":"4","key":"46_CR6","first-page":"692","volume":"39","author":"A Dosovitskiy","year":"2017","unstructured":"Dosovitskiy, A., Springenberg, J., Tatarchenko, M., Brox, T.: Learning to generate chairs, tables and cars with convolutional networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(4), 692\u2013705 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"46_CR7","doi-asserted-by":"crossref","unstructured":"Flynn, J., Neulander, I., Philbin, J., Snavely, N.: Deepstereo: learning to predict new views from the world\u2019s imagery. In: Conference on Computer Vision and Pattern Recognition, pp. 5515\u20135524 (2016)","DOI":"10.1109\/CVPR.2016.595"},{"key":"46_CR8","doi-asserted-by":"crossref","unstructured":"Gadelha, M., Maji, S., Wang, R.: 3D shape induction from 2D views of multiple objects. arXiv preprint arXiv:1612.05872 (2016)","DOI":"10.1109\/3DV.2017.00053"},{"key":"46_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"266","DOI":"10.1007\/978-3-319-49409-8_22","volume-title":"Computer Vision \u2013 ECCV 2016 Workshops","author":"E Grant","year":"2016","unstructured":"Grant, E., Kohli, P., van Gerven, M.: Deep disentangled representations for volumetric reconstruction. In: Hua, G., J\u00e9gou, H. (eds.) ECCV 2016. LNCS, vol. 9915, pp. 266\u2013279. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-49409-8_22"},{"key":"46_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"46_CR11","first-page":"44","volume-title":"Lecture Notes in Computer Science","author":"Geoffrey E. Hinton","year":"2011","unstructured":"Hinton, G., Krizhevsky, A., Wang, S.: Transforming auto-encoders. In: International Conference on Artificial Neural Networks, pp. 44\u201351 (2011)"},{"key":"46_CR12","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Carreira, J., Sminchisescu, C.: Iterated second-order label sensitive pooling for 3D human pose estimation. In: Conference on Computer Vision and Pattern Recognition (2014)","DOI":"10.1109\/CVPR.2014.215"},{"key":"46_CR13","doi-asserted-by":"publisher","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2014","unstructured":"Ionescu, C., Papava, I., Olaru, V., Sminchisescu, C.: Human3.6M: large scale datasets and predictive methods for 3D human sensing in natural environments. IEEE Trans. Pattern Anal. Mach. Intell. 36, 1325\u20131339 (2014)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"46_CR14","doi-asserted-by":"crossref","unstructured":"Joo, H., et al.: Panoptic studio: a massively multiview system for social motion capture. In: International Conference on Computer Vision (2015)","DOI":"10.1109\/ICCV.2015.381"},{"key":"46_CR15","unstructured":"Kar, A., H\u00e4ne, C., Malik, J.: Learning a multi-view stereo machine. In: Advances in Neural Information Processing Systems, pp. 364\u2013375 (2017)"},{"key":"46_CR16","doi-asserted-by":"crossref","unstructured":"Kim, H., Zollh\u00f6fer, M., Tewari, A., Thies, J., Richardt, C., Theobalt, C.: Inversefacenet: deep single-shot inverse face rendering from a single image. arXiv Preprint (2017)","DOI":"10.1109\/CVPR.2018.00486"},{"key":"46_CR17","unstructured":"Kulkarni, T.D., Whitney, W., Kohli, P., Tenenbaum, J.B.: Deep Convolutional Inverse Graphics Network. arXiv (2015)"},{"key":"46_CR18","doi-asserted-by":"crossref","unstructured":"Lassner, C., Pons-Moll, G., Gehler, P.: A generative model of people in clothing. arXiv Preprint (2017)","DOI":"10.1109\/ICCV.2017.98"},{"key":"46_CR19","unstructured":"Ma, L., Jia, X., Sun, Q., Schiele, B., Tuytelaars, T., Gool, L.V.: Pose guided person image generation. In: Advances in Neural Information Processing Systems, pp. 405\u2013415 (2017)"},{"key":"46_CR20","doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.: A simple yet effective baseline for 3D human pose estimation. In: International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.288"},{"key":"46_CR21","doi-asserted-by":"crossref","unstructured":"Mehta, D., et al.: Monocular 3D human pose estimation in the wild using improved CNN supervision. In: International Conference on 3D Vision (2017)","DOI":"10.1109\/3DV.2017.00064"},{"issue":"4","key":"46_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073596","volume":"36","author":"Dushyant Mehta","year":"2017","unstructured":"Mehta, D., et al.: Vnect: real-time 3D human pose estimation with a single RGB camera. In: ACM SIGGRAPH (2017)","journal-title":"ACM Transactions on Graphics"},{"key":"46_CR23","doi-asserted-by":"crossref","unstructured":"Park, E., Yang, J., Yumer, E., Ceylan, D., Berg, A.: Transformation-grounded image generation network for novel 3D view synthesis. In: Conference on Computer Vision and Pattern Recognition, pp. 702\u2013711 (2017)","DOI":"10.1109\/CVPR.2017.82"},{"key":"46_CR24","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhou, X., Derpanis, K., Konstantinos, G., Daniilidis, K.: Coarse-to-fine volumetric prediction for single-image 3D human pose. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.139"},{"key":"46_CR25","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhou, X., Konstantinos, K.D.G., Kostas, D.: Harvesting multiple views for marker-less 3D human pose annotations. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.138"},{"key":"46_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1007\/978-3-319-46448-0_3","volume-title":"Computer Vision \u2013 ECCV 2016","author":"X Peng","year":"2016","unstructured":"Peng, X., Feris, R.S., Wang, X., Metaxas, D.N.: A recurrent encoder-decoder network for sequential face alignment. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 38\u201356. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_3"},{"key":"46_CR27","doi-asserted-by":"crossref","unstructured":"Popa, A.I., Zanfir, M., Sminchisescu, C.: Deep multitask architecture for integrated 2D and 3D human sensing. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.501"},{"key":"46_CR28","unstructured":"Reed, S., Zhang, Y., Zhang, Y., Lee, H.: Deep visual analogy-making. In: Advances in Neural Information Processing Systems, pp. 1252\u20131260 (2015)"},{"key":"46_CR29","unstructured":"Rezende, D., Eslami, S., Mohamed, S., Battaglia, P., Jaderberg, M., Heess, N.: Unsupervised learning of 3D structure from images. In: Advances in Neural Information Processing Systems, pp. 4996\u20135004 (2016)"},{"issue":"6","key":"46_CR30","first-page":"162","volume":"35","author":"H Rhodin","year":"2016","unstructured":"Rhodin, H., et al.: Egocap: egocentric marker-less motion capture with two fisheye cameras. ACM SIGGRAPH Asia 35(6), 162 (2016)","journal-title":"ACM SIGGRAPH Asia"},{"key":"46_CR31","doi-asserted-by":"crossref","unstructured":"Rhodin, H., et al.: Learning monocular 3D human pose estimation from multi-view images. In: Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00880"},{"key":"46_CR32","unstructured":"Rogez, G., Schmid, C.: Mocap guided data augmentation for 3D pose estimation in the wild. In: Advances in Neural Information Processing Systems (2016)"},{"key":"46_CR33","doi-asserted-by":"crossref","unstructured":"Rogez, G., Weinzaepfel, P., Schmid, C.: LCR-Net: localization-classification-regression for human pose. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.134"},{"key":"46_CR34","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Conference on Medical Image Computing and Computer Assisted Intervention (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"46_CR35","doi-asserted-by":"crossref","unstructured":"Shu, Z., Yumer, E., Hadap, S., Sunkavalli, K., Shechtman, E., Samaras, D.: Neural face editing with intrinsic image disentangling. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.578"},{"key":"46_CR36","unstructured":"Tatarchenko, M., Dosovitskiy, A., Brox, T.: Single-view to multi-view: reconstructing unseen views with a convolutional network. CoRR abs\/1511.06702 1, 2 (2015)"},{"key":"46_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"322","DOI":"10.1007\/978-3-319-46478-7_20","volume-title":"Computer Vision \u2013 ECCV 2016","author":"M Tatarchenko","year":"2016","unstructured":"Tatarchenko, M., Dosovitskiy, A., Brox, T.: Multi-view 3D models from single images with a convolutional network. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9911, pp. 322\u2013337. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46478-7_20"},{"key":"46_CR38","doi-asserted-by":"crossref","unstructured":"Tekin, B., M\u00e1rquez-neila, P., Salzmann, M., Fua, P.: Learning to fuse 2D and 3D image cues for monocular body pose estimation. In: International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.425"},{"key":"46_CR39","doi-asserted-by":"crossref","unstructured":"Tewari, A., et al.: Mofa: model-based deep convolutional face autoencoder for unsupervised monocular reconstruction. In: International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.401"},{"key":"46_CR40","unstructured":"Thewlis, J., Bilen, H., Vedaldi, A.: Unsupervised learning of object frames by dense equivariant image labelling. In: Advances in Neural Information Processing Systems, pp. 844\u2013855 (2017)"},{"key":"46_CR41","doi-asserted-by":"crossref","unstructured":"Thewlis, J., Bilen, H., Vedaldi, A.: Unsupervised learning of object landmarks by factorized spatial embeddings. In: International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.348"},{"key":"46_CR42","doi-asserted-by":"crossref","unstructured":"Tome, D., Russell, C., Agapito, L.: Lifting from the deep: convolutional 3D pose estimation from a single image. arXiv preprint, arXiv:1701.00295 (2017)","DOI":"10.1109\/CVPR.2017.603"},{"key":"46_CR43","doi-asserted-by":"crossref","unstructured":"Tran, L., Yin, X., Liu, X.: Disentangled representation learning gan for pose-invariant face recognition. In: CVPR, vol. 3, p. 7 (2017)","DOI":"10.1109\/CVPR.2017.141"},{"key":"46_CR44","doi-asserted-by":"crossref","unstructured":"Tulsiani, S., Efros, A., Malik, J.: Multi-view consistency as supervisory signal for learning shape and pose prediction. arXiv Preprint (2018)","DOI":"10.1109\/CVPR.2018.00306"},{"key":"46_CR45","doi-asserted-by":"crossref","unstructured":"Tulsiani, S., Zhou, T., Efros, A., Malik, J.: Multi-view supervision for single-view reconstruction via differentiable ray consistency. In: Conference on Computer Vision and Pattern Recognition, vol. 1, p. 3 (2017)","DOI":"10.1109\/CVPR.2017.30"},{"key":"46_CR46","doi-asserted-by":"crossref","unstructured":"Tung, H.Y., Harley, A., Seto, W., Fragkiadaki, K.: Adversarial inverse graphics networks: learning 2D-to-3D lifting and image-to-image translation from unpaired supervision. In: The IEEE International Conference on Computer Vision (ICCV), vol. 2 (2017)","DOI":"10.1109\/ICCV.2017.467"},{"key":"46_CR47","unstructured":"Tung, H.Y., Tung, H.W., Yumer, E., Fragkiadaki, K.: Self-supervised learning of motion capture. In: Advances in Neural Information Processing Systems, pp. 5242\u20135252 (2017)"},{"key":"46_CR48","doi-asserted-by":"crossref","unstructured":"Varol, G., et al.: Learning from synthetic humans. In: Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.492"},{"key":"46_CR49","doi-asserted-by":"crossref","unstructured":"Worrall, D., Garbin, S., Turmukhambetov, D., Brostow, G.: Interpretable transformations with encoder-decoder networks. In: International Conference on Computer Vision, vol. 4 (2017)","DOI":"10.1109\/ICCV.2017.611"},{"key":"46_CR50","unstructured":"Yan, X., Yang, J., Yumer, E., Guo, Y., Lee, H.: Perspective transformer nets: learning single-view 3D object reconstruction without 3D supervision. In: Advances in Neural Information Processing Systems, pp. 1696\u20131704 (2016)"},{"key":"46_CR51","unstructured":"Yang, J., Reed, S., Yang, M.H., Lee, H.: Weakly-supervised disentangling with recurrent transformations for 3D view synthesis. In: Advances in Neural Information Processing Systems, pp. 1099\u20131107 (2015)"},{"key":"46_CR52","doi-asserted-by":"crossref","unstructured":"Zhao, B., Wu, X., Cheng, Z.Q., Liu, H., Feng, J.: Multi-view image generation from a single-view. arXiv preprint arXiv:1704.04886 (2017)","DOI":"10.1145\/3240508.3240536"},{"key":"46_CR53","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"286","DOI":"10.1007\/978-3-319-46493-0_18","volume-title":"Computer Vision \u2013 ECCV 2016","author":"T Zhou","year":"2016","unstructured":"Zhou, T., Tulsiani, S., Sun, W., Malik, J., Efros, A.A.: View synthesis by appearance flow. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 286\u2013301. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_18"},{"key":"46_CR54","unstructured":"Zhou, X., Huang, Q., Sun, X., Xue, X., We, Y.: Weakly-supervised transfer for 3D human pose estimation in the wild. arXiv Preprint (2017)"},{"key":"46_CR55","doi-asserted-by":"crossref","unstructured":"Zhou, X., Karpur, A., Gan, C., Luo, L., Huang, Q.: Unsupervised domain adaptation for 3D keypoint prediction from a single depth scan. arXiv preprint arXiv:1712.05765 (2017)","DOI":"10.1007\/978-3-030-01258-8_9"},{"key":"46_CR56","doi-asserted-by":"crossref","unstructured":"Zhu, J.Y., Park, T., Isola, P., Efros, A.: Unpaired image-to-image translation using cycle-consistent adversarial networks. arXiv preprint arXiv:1703.10593 (2017)","DOI":"10.1109\/ICCV.2017.244"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2018"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-01249-6_46","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,5]],"date-time":"2022-10-05T01:01:14Z","timestamp":1664931674000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-01249-6_46"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783030012489","9783030012496"],"references-count":56,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-01249-6_46","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018]]},"assertion":[{"value":"6 October 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Munich","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 September 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 September 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2018.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}