{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T22:29:50Z","timestamp":1784154590077,"version":"3.55.0"},"publisher-location":"Cham","reference-count":101,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732461","type":"print"},{"value":"9783031732478","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73247-8_27","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T12:02:20Z","timestamp":1730376140000},"page":"467-487","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":33,"title":["TRAM: Global Trajectory and\u00a0Motion of\u00a03D Humans from\u00a0in-the-Wild Videos"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9907-8382","authenticated-orcid":false,"given":"Yufu","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9803-7949","authenticated-orcid":false,"given":"Ziyun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4301-1474","authenticated-orcid":false,"given":"Lingjie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0498-0758","authenticated-orcid":false,"given":"Kostas","family":"Daniilidis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"27_CR1","unstructured":"Easymocap - make human motion capture easier. Github (2021). https:\/\/github.com\/zju3dv\/EasyMocap"},{"key":"27_CR2","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: ViViT: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"27_CR3","doi-asserted-by":"crossref","unstructured":"Arnab, A., Doersch, C., Zisserman, A.: Exploiting temporal context for 3D human pose estimation in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3395\u20133404 (2019)","DOI":"10.1109\/CVPR.2019.00351"},{"key":"27_CR4","doi-asserted-by":"crossref","unstructured":"Baradel, F., Br\u00e9gier, R., Groueix, T., Weinzaepfel, P., Kalantidis, Y., Rogez, G.: PoseBERT: a generic transformer module for temporal 3D human modeling. IEEE Trans. Pattern Anal. Mach. Intell. (2022)","DOI":"10.1109\/TPAMI.2022.3216899"},{"key":"27_CR5","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., M\u00fcller, M.: ZoeDepth: zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)"},{"key":"27_CR6","doi-asserted-by":"crossref","unstructured":"Black, M.J., Patel, P., Tesch, J., Yang, J.: Bedlam: a synthetic dataset of bodies exhibiting detailed lifelike animated motion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8726\u20138737 (2023)","DOI":"10.1109\/CVPR52729.2023.00843"},{"key":"27_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"561","DOI":"10.1007\/978-3-319-46454-1_34","volume-title":"Computer Vision \u2013 ECCV 2016","author":"F Bogo","year":"2016","unstructured":"Bogo, F., Kanazawa, A., Lassner, C., Gehler, P., Romero, J., Black, M.J.: Keep it SMPL: automatic estimation of 3D human pose and shape from a single image. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9909, pp. 561\u2013578. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46454-1_34"},{"key":"27_CR8","doi-asserted-by":"crossref","unstructured":"Brachmann, E., et al.: Scene coordinate reconstruction: posing of image collections via incremental learning of a relocalizer. arXiv preprint arXiv:2404.14351 (2024)","DOI":"10.1007\/978-3-031-72992-8_24"},{"issue":"6","key":"27_CR9","doi-asserted-by":"publisher","first-page":"1309","DOI":"10.1109\/TRO.2016.2624754","volume":"32","author":"C Cadena","year":"2016","unstructured":"Cadena, C., et al.: Past, present, and future of simultaneous localization and mapping: toward the robust-perception age. IEEE Trans. Rob. 32(6), 1309\u20131332 (2016)","journal-title":"IEEE Trans. Rob."},{"key":"27_CR10","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"342","DOI":"10.1007\/978-3-031-19769-7_20","volume-title":"ECCV 2022","author":"J Cho","year":"2022","unstructured":"Cho, J., Youwang, K., Oh, T.H.: Cross-attention of disentangled modalities for 3D human mesh recovery with transformers. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13661, pp. 342\u2013359. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19769-7_20"},{"key":"27_CR11","doi-asserted-by":"crossref","unstructured":"Choi, H., Moon, G., Chang, J.Y., Lee, K.M.: Beyond static features for temporally consistent 3D human pose and shape from a video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1964\u20131973 (2021)","DOI":"10.1109\/CVPR46437.2021.00200"},{"key":"27_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"769","DOI":"10.1007\/978-3-030-58571-6_45","volume-title":"Computer Vision \u2013 ECCV 2020","author":"H Choi","year":"2020","unstructured":"Choi, H., Moon, G., Lee, K.M.: Pose2Mesh: graph convolutional network for 3D human pose and mesh recovery from a 2D human pose. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12352, pp. 769\u2013787. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_45"},{"key":"27_CR13","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1007\/978-3-031-20068-7_10","volume-title":"ECCV 2022","author":"V Choutas","year":"2022","unstructured":"Choutas, V., Bogo, F., Shen, J., Valentin, J.: Learning to fit morphable models. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13666, pp. 160\u2013179. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20068-7_10"},{"key":"27_CR14","unstructured":"Dehghani, M., et\u00a0al.: Scaling vision transformers to 22 billion parameters. In: International Conference on Machine Learning, pp. 7480\u20137512. PMLR (2023)"},{"key":"27_CR15","doi-asserted-by":"crossref","unstructured":"Dong, J., Jiang, W., Huang, Q., Bao, H., Zhou, X.: Fast and robust multi-person 3D pose estimation from multiple views. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7792\u20137801 (2019)","DOI":"10.1109\/CVPR.2019.00798"},{"key":"27_CR16","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth $$16 \\times 16$$ words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"27_CR17","doi-asserted-by":"crossref","unstructured":"Goel, S., Pavlakos, G., Rajasegaran, J., Kanazawa, A., Malik, J.: Humans in 4D: reconstructing and tracking humans with transformers. arXiv preprint arXiv:2305.20091 (2023)","DOI":"10.1109\/ICCV51070.2023.01358"},{"key":"27_CR18","doi-asserted-by":"crossref","unstructured":"Guzov, V., Mir, A., Sattler, T., Pons-Moll, G.: Human POSEitioning system (HPS): 3D human pose estimation and self-localization in large scenes from body-mounted sensors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4318\u20134329 (2021)","DOI":"10.1109\/CVPR46437.2021.00430"},{"key":"27_CR19","unstructured":"He, C., Saito, J., Zachary, J., Rushmeier, H., Zhou, Y.: NeMF: neural motion fields for kinematic animation. In: Advances in Neural Information Processing Systems, vol. 35, pp. 4244\u20134256 (2022)"},{"key":"27_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"27_CR21","doi-asserted-by":"crossref","unstructured":"Henning, D.F., Choi, C., Schaefer, S., Leutenegger, S.: BodySLAM++: fast and tightly-coupled visual-inertial camera and human motion tracking. In: 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 3781\u20133788. IEEE (2023)","DOI":"10.1109\/IROS55552.2023.10342291"},{"key":"27_CR22","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"656","DOI":"10.1007\/978-3-031-19842-7_38","volume-title":"ECCV 2022","author":"DF Henning","year":"2022","unstructured":"Henning, D.F., Laidlow, T., Leutenegger, S.: BodySLAM: joint camera localisation, mapping, and human motion tracking. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13699, pp. 656\u2013673. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19842-7_38"},{"key":"27_CR23","doi-asserted-by":"crossref","unstructured":"Hu, M., et al.: Metric3D v2: a versatile monocular geometric foundation model for zero-shot metric depth and surface normal estimation. arXiv preprint arXiv:2404.15506 (2024)","DOI":"10.1109\/TPAMI.2024.3444912"},{"key":"27_CR24","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: Towards accurate marker-less human shape and pose estimation over time. In: 2017 International Conference on 3D Vision (3DV), pp. 421\u2013430. IEEE (2017)","DOI":"10.1109\/3DV.2017.00055"},{"issue":"7","key":"27_CR25","doi-asserted-by":"publisher","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2013","unstructured":"Ionescu, C., Papava, D., Olaru, V., Sminchisescu, C.: Human3.6m: large scale datasets and predictive methods for 3D human sensing in natural environments. IEEE Trans. Pattern Anal. Mach. Intell. 36(7), 1325\u20131339 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"27_CR26","doi-asserted-by":"crossref","unstructured":"Jiang, W., Kolotouros, N., Pavlakos, G., Zhou, X., Daniilidis, K.: Coherent reconstruction of multiple humans from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5579\u20135588 (2020)","DOI":"10.1109\/CVPR42600.2020.00562"},{"key":"27_CR27","doi-asserted-by":"crossref","unstructured":"Joo, H., Neverova, N., Vedaldi, A.: Exemplar fine-tuning for 3D human model fitting towards in-the-wild 3D human pose estimation. In: 2021 International Conference on 3D Vision (3DV), pp. 42\u201352. IEEE (2021)","DOI":"10.1109\/3DV53792.2021.00015"},{"key":"27_CR28","doi-asserted-by":"crossref","unstructured":"Kanazawa, A., Black, M.J., Jacobs, D.W., Malik, J.: End-to-end recovery of human shape and pose. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7122\u20137131 (2018)","DOI":"10.1109\/CVPR.2018.00744"},{"key":"27_CR29","doi-asserted-by":"crossref","unstructured":"Kanazawa, A., Zhang, J.Y., Felsen, P., Malik, J.: Learning 3D human dynamics from video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5614\u20135623 (2019)","DOI":"10.1109\/CVPR.2019.00576"},{"key":"27_CR30","doi-asserted-by":"crossref","unstructured":"Kaufmann, M., et al.: EMDB: the electromagnetic database of global 3D human pose and shape in the wild. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14632\u201314643 (2023)","DOI":"10.1109\/ICCV51070.2023.01345"},{"key":"27_CR31","doi-asserted-by":"crossref","unstructured":"Ke, B., Obukhov, A., Huang, S., Metzger, N., Daudt, R.C., Schindler, K.: Repurposing diffusion-based image generators for monocular depth estimation. arXiv preprint arXiv:2312.02145 (2023)","DOI":"10.1109\/CVPR52733.2024.00907"},{"key":"27_CR32","doi-asserted-by":"crossref","unstructured":"Keller, M., Zuffi, S., Black, M.J., Pujades, S.: OSSO: obtaining skeletal shape from outside. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20492\u201320501 (2022)","DOI":"10.1109\/CVPR52688.2022.01984"},{"key":"27_CR33","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"27_CR34","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Athanasiou, N., Black, M.J.: Vibe: video inference for human body pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5253\u20135263 (2020)","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"27_CR35","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Huang, C.H.P., Hilliges, O., Black, M.J.: PARE: part attention regressor for 3D human body estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11127\u201311137 (2021)","DOI":"10.1109\/ICCV48922.2021.01094"},{"key":"27_CR36","unstructured":"Kocabas, M., et al.: Pace: human and camera motion estimation from in-the-wild videos. arXiv preprint arXiv:2310.13768 (2023)"},{"key":"27_CR37","doi-asserted-by":"crossref","unstructured":"Kolotouros, N., Pavlakos, G., Black, M.J., Daniilidis, K.: Learning to reconstruct 3D human pose and shape via model-fitting in the loop. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2252\u20132261 (2019)","DOI":"10.1109\/ICCV.2019.00234"},{"key":"27_CR38","doi-asserted-by":"crossref","unstructured":"Kolotouros, N., Pavlakos, G., Daniilidis, K.: Convolutional mesh regression for single-image human shape reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4501\u20134510 (2019)","DOI":"10.1109\/CVPR.2019.00463"},{"key":"27_CR39","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"479","DOI":"10.1007\/978-3-031-20065-6_28","volume-title":"ECCV 2022","author":"J Li","year":"2022","unstructured":"Li, J., Bian, S., Xu, C., Liu, G., Yu, G., Lu, C.: D & D: learning human dynamics from dynamic camera. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13665, pp. 479\u2013496. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20065-6_28"},{"key":"27_CR40","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, C., Chen, Z., Bian, S., Yang, L., Lu, C.: HybrIK: a hybrid analytical-neural inverse kinematics solution for 3D human pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3383\u20133393 (2021)","DOI":"10.1109\/CVPR46437.2021.00339"},{"key":"27_CR41","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"590","DOI":"10.1007\/978-3-031-20065-6_34","volume-title":"ECCV 2022","author":"Z Li","year":"2022","unstructured":"Li, Z., Liu, J., Zhang, Z., Xu, S., Yan, Y.: CLIFF: carrying location information in full frames into human pose and shape estimation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13665, pp. 590\u2013606. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20065-6_34"},{"key":"27_CR42","unstructured":"Li, Z., et al.: Train big, then compress: rethinking model size for efficient training and inference of transformers. In: International Conference on Machine Learning, pp. 5958\u20135968. PMLR (2020)"},{"key":"27_CR43","doi-asserted-by":"crossref","unstructured":"Lin, K., Wang, L., Liu, Z.: End-to-end human pose and mesh reconstruction with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1954\u20131963 (2021)","DOI":"10.1109\/CVPR46437.2021.00199"},{"key":"27_CR44","doi-asserted-by":"crossref","unstructured":"Lin, K., Wang, L., Liu, Z.: Mesh graphormer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12939\u201312948 (2021)","DOI":"10.1109\/ICCV48922.2021.01270"},{"key":"27_CR45","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"27_CR46","doi-asserted-by":"crossref","unstructured":"Liu, M., Yang, D., Zhang, Y., Cui, Z., Rehg, J.M., Tang, S.: 4D human body capture from egocentric video via 3D scene grounding. In: 2021 International Conference on 3D Vision (3DV), pp. 930\u2013939. IEEE (2021)","DOI":"10.1109\/3DV53792.2021.00101"},{"key":"27_CR47","doi-asserted-by":"crossref","unstructured":"Loper, M., Mahmood, N., Black, M.J.: Mosh: motion and shape capture from sparse markers. ACM Trans. Graph. 33(6), 220\u20131 (2014)","DOI":"10.1145\/2661229.2661273"},{"issue":"6","key":"27_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2816795.2818013","volume":"34","author":"M Loper","year":"2015","unstructured":"Loper, M., Mahmood, N., Romero, J., Pons-Moll, G., Black, M.J.: SMPL: a skinned multi-person linear model. ACM Trans. Graph. (TOG) 34(6), 1\u201316 (2015)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"27_CR49","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"27_CR50","doi-asserted-by":"crossref","unstructured":"Luo, Z., Golestaneh, S.A., Kitani, K.M.: 3D human motion estimation via motion compression and refinement. In: Proceedings of the Asian Conference on Computer Vision (2020)","DOI":"10.1007\/978-3-030-69541-5_20"},{"key":"27_CR51","doi-asserted-by":"crossref","unstructured":"Mahmood, N., Ghorbani, N., Troje, N.F., Pons-Moll, G., Black, M.J.: AMASS: archive of motion capture as surface shapes. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5442\u20135451 (2019)","DOI":"10.1109\/ICCV.2019.00554"},{"key":"27_CR52","doi-asserted-by":"crossref","unstructured":"Mehta, D., et al.: Monocular 3D human pose estimation in the wild using improved CNN supervision. In: 2017 International Conference on 3D Vision (3DV), pp. 506\u2013516. IEEE (2017)","DOI":"10.1109\/3DV.2017.00064"},{"key":"27_CR53","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"752","DOI":"10.1007\/978-3-030-58571-6_44","volume-title":"Computer Vision \u2013 ECCV 2020","author":"G Moon","year":"2020","unstructured":"Moon, G., Lee, K.M.: I2L-MeshNet: image-to-lixel prediction network for accurate 3D human pose and mesh estimation from a single RGB image. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12352, pp. 752\u2013768. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_44"},{"issue":"5","key":"27_CR54","doi-asserted-by":"publisher","first-page":"1255","DOI":"10.1109\/TRO.2017.2705103","volume":"33","author":"R Mur-Artal","year":"2017","unstructured":"Mur-Artal, R., Tard\u00f3s, J.D.: ORB-SLAM2: an open-source SLAM system for monocular, stereo, and RGB-D cameras. IEEE Trans. Rob. 33(5), 1255\u20131262 (2017)","journal-title":"IEEE Trans. Rob."},{"key":"27_CR55","doi-asserted-by":"crossref","unstructured":"Omran, M., Lassner, C., Pons-Moll, G., Gehler, P., Schiele, B.: Neural body fitting: unifying deep learning and model based human pose and shape estimation. In: 2018 International Conference on 3D Vision (3DV), pp. 484\u2013494. IEEE (2018)","DOI":"10.1109\/3DV.2018.00062"},{"key":"27_CR56","doi-asserted-by":"crossref","unstructured":"Pan, B., et al.: Copilot: human collision prediction and localization from multi-view egocentric videos. arXiv preprint arXiv:2210.01781 (2022)","DOI":"10.1109\/ICCV51070.2023.00485"},{"key":"27_CR57","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1007\/s11263-015-0804-2","volume":"115","author":"HS Park","year":"2015","unstructured":"Park, H.S., Shiratori, T., Matthews, I., Sheikh, Y.: 3D trajectory reconstruction under perspective projection. Int. J. Comput. Vis. 115, 115\u2013135 (2015)","journal-title":"Int. J. Comput. Vis."},{"key":"27_CR58","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., et al.: Expressive body capture: 3D hands, face, and body from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10975\u201310985 (2019)","DOI":"10.1109\/CVPR.2019.01123"},{"key":"27_CR59","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Malik, J., Kanazawa, A.: Human mesh recovery from multiple shots. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1485\u20131495 (2022)","DOI":"10.1109\/CVPR52688.2022.00154"},{"key":"27_CR60","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"732","DOI":"10.1007\/978-3-031-19836-6_41","volume-title":"ECCV 2022","author":"G Pavlakos","year":"2022","unstructured":"Pavlakos, G., Weber, E., Tancik, M., Kanazawa, A.: The one where they reconstructed 3D humans and environments in TV shows. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13697, pp. 732\u2013749. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19836-6_41"},{"issue":"6","key":"27_CR61","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3272127.3275014","volume":"37","author":"XB Peng","year":"2018","unstructured":"Peng, X.B., Kanazawa, A., Malik, J., Abbeel, P., Levine, S.: SFV: reinforcement learning of physical skills from videos. ACM Trans. Graph. (TOG) 37(6), 1\u201314 (2018)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"27_CR62","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Action-conditioned 3D human motion synthesis with transformer VAE. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10985\u201310995 (2021)","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"27_CR63","doi-asserted-by":"crossref","unstructured":"Rempe, D., Birdal, T., Hertzmann, A., Yang, J., Sridhar, S., Guibas, L.J.: Humor: 3D human motion model for robust pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11488\u201311499 (2021)","DOI":"10.1109\/ICCV48922.2021.01129"},{"key":"27_CR64","doi-asserted-by":"crossref","unstructured":"Rempe, D., et al.: Trace and pace: controllable pedestrian animation via guided trajectory diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13756\u201313766 (2023)","DOI":"10.1109\/CVPR52729.2023.01322"},{"key":"27_CR65","unstructured":"Romero, J., Tzionas, D., Black, M.J.: Embodied hands: modeling and capturing hands and bodies together. arXiv preprint arXiv:2201.02610 (2022)"},{"key":"27_CR66","doi-asserted-by":"crossref","unstructured":"Rong, Y., Shiratori, T., Joo, H.: FrankMocap: Fast monocular 3D hand and body motion capture by regression and integration. arXiv preprint arXiv:2008.08324 (2020)","DOI":"10.1109\/ICCVW54120.2021.00201"},{"key":"27_CR67","doi-asserted-by":"crossref","unstructured":"Saini, N., Huang, C.H.P., Black, M.J., Ahmad, A.: SmartMocap: joint estimation of human and camera motion using uncalibrated RGB cameras. IEEE Robot. Autom. Lett. (2023)","DOI":"10.1109\/LRA.2023.3264743"},{"key":"27_CR68","doi-asserted-by":"crossref","unstructured":"Schonberger, J.L., Frahm, J.M.: Structure-from-motion revisited. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4104\u20134113 (2016)","DOI":"10.1109\/CVPR.2016.445"},{"key":"27_CR69","doi-asserted-by":"crossref","unstructured":"Shen, X., Yang, Z., Wang, X., Ma, J., Zhou, C., Yang, Y.: Global-to-local modeling for video-based 3D human pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8887\u20138896 (2023)","DOI":"10.1109\/CVPR52729.2023.00858"},{"key":"27_CR70","doi-asserted-by":"crossref","unstructured":"Shin, S., Kim, J., Halilaj, E., Black, M.J.: WHAM: reconstructing world-grounded humans with accurate 3D motion. arXiv preprint arXiv:2312.07531 (2023)","DOI":"10.1109\/CVPR52733.2024.00202"},{"key":"27_CR71","unstructured":"Sminchisescu, C., Telea, A.C.: Human pose estimation from silhouettes. A consistent approach using distance level sets. In: 10th International Conference on Computer Graphics, Visualization and Computer Vision (WSCG 2002), vol.\u00a010 (2002)"},{"key":"27_CR72","unstructured":"Smith, C., Charatan, D., Tewari, A., Sitzmann, V.: FlowMap: high-quality camera poses, intrinsics, and depth via gradient descent. arXiv preprint arXiv:2404.15259 (2024)"},{"key":"27_CR73","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"744","DOI":"10.1007\/978-3-030-58565-5_44","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Song","year":"2020","unstructured":"Song, J., Chen, X., Hilliges, O.: Human body model fitting by learned gradient descent. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12365, pp. 744\u2013760. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58565-5_44"},{"issue":"4","key":"27_CR74","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3450626.3459881","volume":"40","author":"S Starke","year":"2021","unstructured":"Starke, S., Zhao, Y., Zinno, F., Komura, T.: Neural animation layering for synthesizing martial arts movements. ACM Trans. Graph. (TOG) 40(4), 1\u201316 (2021)","journal-title":"ACM Trans. Graph. (TOG)"},{"issue":"21","key":"27_CR75","doi-asserted-by":"publisher","first-page":"7315","DOI":"10.3390\/s21217315","volume":"21","author":"J Stenum","year":"2021","unstructured":"Stenum, J., Cherry-Allen, K.M., Pyles, C.O., Reetzke, R.D., Vignos, M.F., Roemmich, R.T.: Applications of pose estimation in human health and performance across the lifespan. Sensors 21(21), 7315 (2021)","journal-title":"Sensors"},{"key":"27_CR76","doi-asserted-by":"crossref","unstructured":"Sturm, J., Engelhard, N., Endres, F., Burgard, W., Cremers, D.: A benchmark for the evaluation of RGB-D SLAM systems. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 573\u2013580. IEEE (2012)","DOI":"10.1109\/IROS.2012.6385773"},{"key":"27_CR77","doi-asserted-by":"crossref","unstructured":"Sun, Y., Bao, Q., Liu, W., Fu, Y., Black, M.J., Mei, T.: Monocular, one-stage, regression of multiple 3D people. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11179\u201311188 (2021)","DOI":"10.1109\/ICCV48922.2021.01099"},{"key":"27_CR78","doi-asserted-by":"crossref","unstructured":"Sun, Y., Bao, Q., Liu, W., Mei, T., Black, M.J.: Trace: 5D temporal regression of avatars with dynamic cameras in 3D environments. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8856\u20138866 (2023)","DOI":"10.1109\/CVPR52729.2023.00855"},{"key":"27_CR79","doi-asserted-by":"crossref","unstructured":"Sun, Y., Liu, W., Bao, Q., Fu, Y., Mei, T., Black, M.J.: Putting people in their place: monocular regression of 3D people in depth. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13243\u201313252 (2022)","DOI":"10.1109\/CVPR52688.2022.01289"},{"key":"27_CR80","unstructured":"Teed, Z., Deng, J.: Droid-SLAM: deep visual SLAM for monocular, stereo, and RGB-D cameras. In: Advances in Neural Information Processing Systems, vol. 34, pp. 16558\u201316569 (2021)"},{"key":"27_CR81","doi-asserted-by":"crossref","unstructured":"Ugrinovic, N., et al.: MultiPhys: multi-person physics-aware 3D motion estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2331\u20132340 (2024)","DOI":"10.1109\/CVPR52733.2024.00226"},{"key":"27_CR82","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"27_CR83","doi-asserted-by":"crossref","unstructured":"Von\u00a0Marcard, T., Henschel, R., Black, M.J., Rosenhahn, B., Pons-Moll, G.: Recovering accurate 3D human pose in the wild using IMUs and a moving camera. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 601\u2013617 (2018)","DOI":"10.1007\/978-3-030-01249-6_37"},{"key":"27_CR84","doi-asserted-by":"crossref","unstructured":"Wan, Z., Li, Z., Tian, M., Liu, J., Yi, S., Li, H.: Encoder-decoder with multi-level attention for 3D human shape and pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13033\u201313042 (2021)","DOI":"10.1109\/ICCV48922.2021.01279"},{"key":"27_CR85","doi-asserted-by":"crossref","unstructured":"Wang, C.Y., Bochkovskiy, A., Liao, H.Y.M.: YOLOv7: trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7464\u20137475 (2023)","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"27_CR86","doi-asserted-by":"crossref","unstructured":"Wang, J., Liu, L., Xu, W., Sarkar, K., Theobalt, C.: Estimating egocentric 3D human pose in global space. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11500\u201311509 (2021)","DOI":"10.1109\/ICCV48922.2021.01130"},{"key":"27_CR87","doi-asserted-by":"crossref","unstructured":"Wang, J., Luvizon, D., Xu, W., Liu, L., Sarkar, K., Theobalt, C.: Scene-aware egocentric 3D human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13031\u201313040 (2023)","DOI":"10.1109\/CVPR52729.2023.01252"},{"key":"27_CR88","doi-asserted-by":"crossref","unstructured":"Wang, Y., Daniilidis, K.: Refit: recurrent fitting network for 3D human recovery. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14644\u201314654 (2023)","DOI":"10.1109\/ICCV51070.2023.01346"},{"key":"27_CR89","doi-asserted-by":"crossref","unstructured":"Wei, W.L., Lin, J.C., Liu, T.L., Liao, H.Y.M.: Capturing humans in motion: temporal-attentive 3D human pose and shape estimation from monocular video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13211\u201313220 (2022)","DOI":"10.1109\/CVPR52688.2022.01286"},{"key":"27_CR90","doi-asserted-by":"crossref","unstructured":"Weiss, A., Hirshberg, D., Black, M.J.: Home 3D body scans from noisy image and range data. In: 2011 International Conference on Computer Vision, pp. 1951\u20131958. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126465"},{"key":"27_CR91","doi-asserted-by":"crossref","unstructured":"Xiang, D., Joo, H., Sheikh, Y.: Monocular total capture: posing face, body, and hands in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10965\u201310974 (2019)","DOI":"10.1109\/CVPR.2019.01122"},{"key":"27_CR92","doi-asserted-by":"crossref","unstructured":"Xu, H., Bazavan, E.G., Zanfir, A., Freeman, W.T., Sukthankar, R., Sminchisescu, C.: GHUM & GHUML: generative 3D human shape and articulated pose models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6184\u20136193 (2020)","DOI":"10.1109\/CVPR42600.2020.00622"},{"key":"27_CR93","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: ViTPose: simple vision transformer baselines for human pose estimation. In: Advances in Neural Information Processing Systems, vol. 35, pp. 38571\u201338584 (2022)"},{"key":"27_CR94","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Xu, X., Feng, J., Zhao, H.: Depth anything: unleashing the power of large-scale unlabeled data. arXiv preprint arXiv:2401.10891 (2024)","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"27_CR95","doi-asserted-by":"crossref","unstructured":"Ye, V., Pavlakos, G., Malik, J., Kanazawa, A.: Decoupling human and camera motion from videos in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21222\u201321232 (2023)","DOI":"10.1109\/CVPR52729.2023.02033"},{"key":"27_CR96","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Iqbal, U., Molchanov, P., Kitani, K., Kautz, J.: GLAMR: global occlusion-aware human mesh recovery with dynamic cameras. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11038\u201311049 (2022)","DOI":"10.1109\/CVPR52688.2022.01076"},{"key":"27_CR97","doi-asserted-by":"crossref","unstructured":"Zanfir, A., Bazavan, E.G., Zanfir, M., Freeman, W.T., Sukthankar, R., Sminchisescu, C.: Neural descent for visual 3D human pose and shape. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14484\u201314493 (2021)","DOI":"10.1109\/CVPR46437.2021.01425"},{"key":"27_CR98","doi-asserted-by":"crossref","unstructured":"Zhai, X., Kolesnikov, A., Houlsby, N., Beyer, L.: Scaling vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12104\u201312113 (2022)","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"27_CR99","doi-asserted-by":"crossref","unstructured":"Zhang, H., et al.: PyMAF: 3D human pose and shape regression with pyramidal mesh alignment feedback loop. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11446\u201311456 (2021)","DOI":"10.1109\/ICCV48922.2021.01125"},{"key":"27_CR100","doi-asserted-by":"crossref","unstructured":"Zhang, S., Ma, Q., Zhang, Y., Aliakbarian, S., Cosker, D., Tang, S.: Probabilistic human mesh recovery in 3D scenes from egocentric views. arXiv preprint arXiv:2304.06024 (2023)","DOI":"10.1109\/ICCV51070.2023.00734"},{"key":"27_CR101","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"180","DOI":"10.1007\/978-3-031-20068-7_11","volume-title":"ECCV 2022","author":"S Zhang","year":"2022","unstructured":"Zhang, S., et al.: EgoBody: human body shape and motion of interacting people from head-mounted devices. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13666, pp. 180\u2013200. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20068-7_11"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73247-8_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T12:12:20Z","timestamp":1730376740000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73247-8_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031732461","9783031732478"],"references-count":101,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73247-8_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}