{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T01:28:07Z","timestamp":1764811687819,"version":"build-2065373602"},"reference-count":31,"publisher":"Tsinghua University Press","issue":"1","license":[{"start":{"date-parts":[[2023,3,1]],"date-time":"2023-03-01T00:00:00Z","timestamp":1677628800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2022,10,18]],"date-time":"2022-10-18T00:00:00Z","timestamp":1666051200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Comp. Visual. Med."],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1007\/s41095-022-0272-x","type":"journal-article","created":{"date-parts":[[2022,10,18]],"date-time":"2022-10-18T03:02:42Z","timestamp":1666062162000},"page":"123-139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Imposing temporal consistency on deep monocular body shape and pose estimation"],"prefix":"10.26599","volume":"9","author":[{"given":"Alexandra","family":"Zimmer","sequence":"first","affiliation":[{"name":"Fraunhofer Heinrich-Hertz-Institut, 10587 Berlin, Germany; Technische Universit&#x00E4;t Berlin, 10623 Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anna","family":"Hilsmann","sequence":"additional","affiliation":[{"name":"Fraunhofer Heinrich-Hertz-Institut, 10587 Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wieland","family":"Morgenstern","sequence":"additional","affiliation":[{"name":"Fraunhofer Heinrich-Hertz-Institut, 10587 Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peter","family":"Eisert","sequence":"additional","affiliation":[{"name":"Fraunhofer Heinrich-Hertz-Institut, 10587 Berlin, Germany; Humboldt Universit&#x00E4;t zu Berlin, 10117 Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"11138","reference":[{"key":"272_CR1","doi-asserted-by":"crossref","unstructured":"Pavlakos, G.; Choutas, V.; Ghorbani, N.; Bolkart, T.; Osman, A. A.; Tzionas, D.; Black, M. J. Expressive body capture: 3D hands, face, and body from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 10967\u201310977, 2019.","DOI":"10.1109\/CVPR.2019.01123"},{"issue":"1","key":"272_CR2","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","volume":"43","author":"Z Cao","year":"2021","unstructured":"Cao, Z.; Hidalgo, G.; Simon, T.; Wei, S.-E.; Sheikh, Y. OpenPose: Realtime multi-person 2D pose estimation using part affinity fields. IEEE Transactions on Pattern Analysis and Machine Intelligence Vol. 43, No. 1, 172\u2013186, 2021.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"272_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"561","DOI":"10.1007\/978-3-319-46454-1_34","volume-title":"Computer Vision \u2014 ECCV 2016","author":"F Bogo","year":"2016","unstructured":"Bogo, F.; Kanazawa, A.; Lassner, C.; Gehler, P.; Romero, J.; Black, M. J. Keep it SMPL: Automatic estimation of 3D human pose and shape from a single image. In: Computer Vision \u2014 ECCV 2016. Lecture Notes in Computer Science, Vol. 9909. Leibe, B.; Matas, J.; Sebe, N.; Welling, M. Eds. Springer Cham, 561\u2013578, 2016."},{"key":"272_CR4","doi-asserted-by":"crossref","unstructured":"Loper, M.; Mahmood, N.; Romero, J.; Pons-Moll, G.; Black, M. J. SMPL: A skinned multi-person linear model. ACM Transactions on Graphics Vol. 34, No. 6, Article No. 248, 2015.","DOI":"10.1145\/2816795.2818013"},{"key":"272_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1007\/978-3-030-58607-2_2","volume-title":"Computer Vision \u2014 ECCV 2020","author":"V Choutas","year":"2020","unstructured":"Choutas, V.; Pavlakos, G.; Bolkart, T.; Tzionas, D.; Black, M. J. Monocular expressive body regression through body-driven attention. In: Computer Vision \u2014 ECCV 2020. Lecture Notes in Computer Science, Vol. 12355. Vedaldi, A.; Bischof, H.; Brox, T.; Frahm, J. M. Eds. Springer Cham, 20\u201340, 2020."},{"key":"272_CR6","doi-asserted-by":"crossref","unstructured":"Kanazawa, A.; Black, M. J.; Jacobs, D. W.; Malik, J. End-to-end recovery of human shape and pose. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 7122\u20137131, 2018.","DOI":"10.1109\/CVPR.2018.00744"},{"key":"272_CR7","doi-asserted-by":"crossref","unstructured":"Zhou, Y. X.; Habermann, M.; Habibie, I.; Tewari, A.; Theobalt, C.; Xu, F. Monocular real-time full body capture with inter-part correlations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4809\u20134820, 2021.","DOI":"10.1109\/CVPR46437.2021.00478"},{"key":"272_CR8","doi-asserted-by":"crossref","unstructured":"Romero, J.; Tzionas, D.; Black, M. J. Embodied hands: Modeling and capturing hands and bodies together. ACM Transactions on Graphics Vol. 36, No. 6, Article No. 245, 2017.","DOI":"10.1145\/3130800.3130883"},{"key":"272_CR9","doi-asserted-by":"crossref","unstructured":"Zhang, H.; Tian, Y.; Zhou, X.; Ouyang, W.; Liu, Y.; Wang, L.; Sun, Z. PyMAF: 3D human pose and shape regression with pyramidal mesh alignment feedback loop. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 11446\u201311456, 2021.","DOI":"10.1109\/ICCV48922.2021.01125"},{"key":"272_CR10","doi-asserted-by":"crossref","unstructured":"Lassner, C.; Romero, J.; Kiefel, M.; Bogo, F.; Black, M. J.; Gehler, P. V. Unite the people: Closing the loop between 3D and 2D human representations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 4704\u20134713, 2017.","DOI":"10.1109\/CVPR.2017.500"},{"key":"272_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/978-3-030-58610-2_3","volume-title":"Computer Vision \u2014 ECCV 2020","author":"J Y Zhang","year":"2020","unstructured":"Zhang, J. Y.; Pepose, S.; Joo, H.; Ramanan, D.; Malik, J.; Kanazawa, A. Perceiving 3D human-object spatial arrangements from a single image in the wild. In: Computer Vision \u2014 ECCV 2020. Lecture Notes in Computer Science, Vol. 12357. Vedaldi, A.; Bischof, H.; Brox, T.; Frahm, J. M. Eds. Springer Cham, 34\u201351, 2020."},{"key":"272_CR12","doi-asserted-by":"publisher","unstructured":"Xu, X. Y.; Chen, H.; Moreno-Noguer, F.; Jeni, L. A.; de la Torre, F. 3D human pose, shape and texture from low-resolution images and videos. IEEE Transactions on Pattern Analysis and Machine Intelligence doi: https:\/\/doi.org\/10.1109\/TPAMI.2021.3070002, 2021.","DOI":"10.1109\/TPAMI.2021.3070002"},{"key":"272_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1007\/978-3-030-58568-6_21","volume-title":"Computer Vision \u2014 ECCV 2020","author":"S H Zou","year":"2020","unstructured":"Zou, S. H.; Zuo, X. X.; Qian, Y. M.; Wang, S.; Xu, C.; Gong, M. L.; Cheng, L. 3D human shape reconstruction from a polarization image. In: Computer Vision \u2014 ECCV 2020. Lecture Notes in Computer Science, Vol. 12359. Vedaldi, A.; Bischof, H.; Brox, T.; Frahm, J. M. Eds. Springer Cham, 351\u2013368, 2020."},{"key":"272_CR14","doi-asserted-by":"publisher","first-page":"1617","DOI":"10.1109\/TMM.2020.3001506","volume":"23","author":"X X Zuo","year":"2021","unstructured":"Zuo, X. X.; Wang, S.; Zheng, J. B.; Yu, W. W.; Gong, M. L.; Yang, R. G.; Cheng, L. SparseFusion: Dynamic human avatar modeling from sparse RGBD images. IEEE Transactions on Multimedia Vol. 23, 1617\u20131629, 2021.","journal-title":"IEEE Transactions on Multimedia"},{"key":"272_CR15","doi-asserted-by":"crossref","unstructured":"Xu, L.; Xu, W. P.; Golyanik, V.; Habermann, M.; Fang, L.; Theobalt, C. EventCap: Monocular 3D capture of high-speed human motions using an event camera. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4967\u20134977, 2020.","DOI":"10.1109\/CVPR42600.2020.00502"},{"key":"272_CR16","doi-asserted-by":"crossref","unstructured":"Kocabas, M.; Athanasiou, N.; Black, M. J. VIBE: Video inference for human body pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 5252\u20135262, 2020.","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"272_CR17","doi-asserted-by":"crossref","unstructured":"Mahmood, N.; Ghorbani, N.; Troje, N. F.; Pons-Moll, G.; Black, M. AMASS: Archive of motion capture as surface shapes. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 5441\u20135450, 2019.","DOI":"10.1109\/ICCV.2019.00554"},{"key":"272_CR18","doi-asserted-by":"crossref","unstructured":"Xiang, D. L.; Joo, H.; Sheikh, Y. Monocular total capture: Posing face, body, and hands in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 10957\u201310966, 2019.","DOI":"10.1109\/CVPR.2019.01122"},{"key":"272_CR19","doi-asserted-by":"crossref","unstructured":"Joo, H.; Simon, T.; Sheikh, Y. Total capture: A 3D deformation model for tracking faces, hands, and bodies. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8320\u20138329, 2018.","DOI":"10.1109\/CVPR.2018.00868"},{"key":"272_CR20","unstructured":"Hodgins, J. K. CMU Graphics Lab Motion Capture Database. Available at http:\/\/mocap.cs.cmu.edu\/resources.php."},{"key":"272_CR21","doi-asserted-by":"crossref","unstructured":"Rong, Y.; Shiratori, T.; Joo, H. FrankMocap: Fast monocular 3D hand and body motion capture by regression and integration. arXiv preprint arXiv: 2008.08324, 2020.","DOI":"10.1109\/ICCVW54120.2021.00201"},{"key":"272_CR22","doi-asserted-by":"crossref","unstructured":"Kolotouros, N.; Pavlakos, G.; Black, M.; Daniilidis, K. Learning to reconstruct 3D human pose and shape via model-fitting in the loop. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2252\u20132261, 2019.","DOI":"10.1109\/ICCV.2019.00234"},{"key":"272_CR23","doi-asserted-by":"crossref","unstructured":"Caliskan, A.; Mustafa, A.; Hilton, A. Temporal consistency loss for high resolution textured and clothed 3D human reconstruction from monocular video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 1780\u20131790, 2021.","DOI":"10.1109\/CVPRW53098.2021.00197"},{"key":"272_CR24","doi-asserted-by":"crossref","unstructured":"He, Y. N.; Pang, A. Q.; Chen, X.; Liang, H.; Wu, M. Y.; Ma, Y. X.; Xu, L. ChallenCap: Monocular 3D capture of challenging human performances using multi-modal references. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 11395\u201311406, 2021.","DOI":"10.1109\/CVPR46437.2021.01124"},{"key":"272_CR25","unstructured":"Zheng, C.; Wu, W. H.; Chen, C.; Yang, T.; Zhu, S. J.; Shen, J.; Kehtarnavaz, N.; Shah, M. Deep learning-based human pose estimation: A survey. arXiv preprint arXiv:2012.13392, 2020."},{"key":"272_CR26","doi-asserted-by":"crossref","unstructured":"Pavllo, D.; Feichtenhofer, C.; Grangier, D.; Auli, M. 3D human pose estimation in video with temporal convolutions and semi-supervised training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 7745\u20137754, 2019.","DOI":"10.1109\/CVPR.2019.00794"},{"issue":"01","key":"272_CR27","doi-asserted-by":"publisher","first-page":"8989","DOI":"10.1609\/aaai.v33i01.33018989","volume":"33","author":"Y-H Wen","year":"2019","unstructured":"Wen, Y.-H.; Gao, L.; Fu, H.; Zhang, F.-L.; Xia, S. Graph CNNs with motif and variable temporal block for skeleton-based action recognition. Proceedings of the AAAI Conference on Artificial Intelligence Vol. 33, No. 01, 8989\u20138996, 2019.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"272_CR28","unstructured":"Teschner, M.; Kimmerle, S.; Heidelberger, B.; Zachmann, G.; Raghupathi, L.; Fuhrmann, A.; Cani, M.; Faure, F.; Magnenat-Thalmann, N.; Stra\u00dfer, W.; et al. Collision detection for deformable objects. In: Proceedings of the 25th Annual Conference of the European Association for Computer Graphics, Eurographics 2004 \u2014 State of the Art Reports, 2004."},{"key":"272_CR29","series-title":"Springer Series in Operations Research and Financial Engineering","first-page":"276","volume-title":"Numerical Optimization","author":"J Nocedal","year":"2006","unstructured":"Nocedal, J.; Wright, S. J. Nonlinear equations. In: Numerical Optimization. Springer Series in Operations Research and Financial Engineering. Springer New York, 276\u2013312, 2006."},{"issue":"6","key":"272_CR30","doi-asserted-by":"publisher","first-page":"e0253157","DOI":"10.1371\/journal.pone.0253157","volume":"16","author":"S Ghorbani","year":"2021","unstructured":"Ghorbani, S.; Mahdaviani, K.; Thaler, A.; Kording, K.; Cook, D. J.; Blohm, G.; Troje, N. F. MoVi: A large multi-purpose human motion and video dataset. PLoS ONE Vol. 16, No. 6, e0253157, 2021.","journal-title":"PLoS ONE"},{"key":"272_CR31","doi-asserted-by":"crossref","unstructured":"Fournier, M.; Dischler, J. M.; Bechmann, D. 3D distance transform adaptive filtering for smoothing and denoising triangle meshes. In: Proceedings of the 4th International Conference on Computer Graphics and Interactive Techniques in Australasia and Southeast Asia, 407\u2013416, 2006.","DOI":"10.1145\/1174429.1174497"}],"container-title":["Computational Visual Media"],"original-title":[],"link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-022-0272-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41095-022-0272-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-022-0272-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10750449\/10897681\/10897690.pdf?arnumber=10897690","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T18:38:54Z","timestamp":1762367934000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10897690\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3]]},"references-count":31,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1007\/s41095-022-0272-x","relation":{},"ISSN":["2096-0662","2096-0433"],"issn-type":[{"type":"electronic","value":"2096-0662"},{"type":"print","value":"2096-0433"}],"subject":[],"published":{"date-parts":[[2023,3]]},"assertion":[{"value":"6 October 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 January 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 October 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration of competing interest"}}]}}