{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:07:33Z","timestamp":1784268453130,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763951","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["PhysHMR: Learning Humanoid Control Policies from Vision for Physically Plausible Human Motion Reconstruction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0625-651X","authenticated-orcid":false,"given":"Qiao","family":"Feng","sequence":"first","affiliation":[{"name":"University of Pennsylvania, Philadelphia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4001-0630","authenticated-orcid":false,"given":"Yiming","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Pennsylvania, Philadelphia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9907-8382","authenticated-orcid":false,"given":"Yufu","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Pennsylvania, Philadelphia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3578-2711","authenticated-orcid":false,"given":"Jiatao","family":"Gu","sequence":"additional","affiliation":[{"name":"University of Pennsylvania, Philadelphia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4301-1474","authenticated-orcid":false,"given":"Lingjie","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Pennsylvania, Philadelphia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00351"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_34"},{"key":"e_1_3_3_2_4_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track","author":"Cai Zhongang","year":"2023","unstructured":"Zhongang Cai, Wanqi Yin, Ailing Zeng, Chen Wei, Qingping Sun, Yanjun Wang, Hui\u00a0En Pang, Haiyi Mei, Mingyuan Zhang, Lei Zhang, Chen\u00a0Change Loy, Lei Yang, and Ziwei Liu. 2023. SMPLer-X: Scaling Up Expressive Human Pose and Shape Estimation. In Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track."},{"key":"e_1_3_3_2_5_1","volume-title":"SIGGRAPH Asia Conference Papers (SA Conference Papers)","author":"Dou Zhiyang","year":"2023","unstructured":"Zhiyang Dou, Xuelin Chen, Qingnan Fan, Taku Komura, and Wenping Wang. 2023. C\u00b7ASE: Learning Conditional Adversarial Skill Embeddings for Physics-based Characters. In SIGGRAPH Asia Conference Papers (SA Conference Papers)."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01284"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01358"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2017.00055"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"crossref","unstructured":"Catalin Ionescu Dragos Papava Vlad Olaru and Cristian Sminchisescu. 2014. Human3.6M: Large Scale Datasets and Predictive Methods for 3D Human Sensing in Natural Environments. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI) (2014).","DOI":"10.1109\/TPAMI.2013.248"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618175"},{"key":"e_1_3_3_2_11_1","unstructured":"Glenn Jocher Ayush Chaurasia and Jing Qiu. 2023. Ultralytics YOLOv8. https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01345"},{"key":"e_1_3_3_2_13_1","unstructured":"Makito Kobayashi Chen-Chieh Liao Keito Inoue Sentaro Yojima and Masafumi Takahashi. 2023. Motion Capture Dataset for Practical Use of AI-based Motion Editing and Stylization. arxiv:https:\/\/arXiv.org\/abs\/2306.08861\u00a0[cs.CV]"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20065-6_28"},{"key":"e_1_3_3_2_15_1","unstructured":"Ruilong Li Shan Yang David\u00a0A. Ross and Angjoo Kanazawa. 2021. Learn to Dance with AIST++: Music Conditioned 3D Dance Generation. arxiv:https:\/\/arXiv.org\/abs\/2101.08779\u00a0[cs.CV]"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20065-6_34"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"crossref","unstructured":"Matthew Loper Naureen Mahmood Javier Romero Gerard Pons-Moll and Michael\u00a0J. Black. 2015. SMPL: A Skinned Multi-Person Linear Model. ACM Transactions on Graphics (TOG) (2015).","DOI":"10.1145\/2816795.2818013"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00061"},{"key":"e_1_3_3_2_19_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Luo Zhengyi","year":"2024","unstructured":"Zhengyi Luo, Jinkun Cao, Josh Merel, Alexander Winkler, Jing Huang, Kris\u00a0M. Kitani, and Weipeng Xu. 2024b. Universal Humanoid Motion Representations for Physics-Based Control. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01000"},{"key":"e_1_3_3_2_21_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Luo Zhengyi","year":"2022","unstructured":"Zhengyi Luo, Shun Iwase, Ye Yuan, and Kris Kitani. 2022. Embodied Scene-aware Human Pose Estimation. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00554"},{"key":"e_1_3_3_2_23_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track","author":"Makoviychuk Viktor","year":"2021","unstructured":"Viktor Makoviychuk, Lukasz Wawrzyniak, Yunrong Guo, Michelle Lu, Kier Storey, Miles Macklin, David Hoeller, Nikita Rudin, Arthur Allshire, Ankur Handa, and Gavriel State. 2021. Isaac Gym: High Performance GPU Based Physics Simulation For Robot Learning. In Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track."},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58539-6_36"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01123"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00894"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"crossref","unstructured":"Xue\u00a0Bin Peng Pieter Abbeel Sergey Levine and Michiel van\u00a0de Panne. 2018a. DeepMimic: Example-guided Deep Reinforcement Learning of Physics-based Character Skills. ACM Transactions on Graphics (TOG) (2018).","DOI":"10.1145\/3197517.3201311"},{"key":"e_1_3_3_2_28_1","unstructured":"Xue\u00a0Bin Peng Yunrong Guo Lina Halper Sergey Levine and Sanja Fidler. 2022. ASE: Large-Scale Reusable Adversarial Skill Embeddings for Physically Simulated Characters. ACM Transactions on Graphics (TOG) (2022)."},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"crossref","unstructured":"Xue\u00a0Bin Peng Angjoo Kanazawa Jitendra Malik Pieter Abbeel and Sergey Levine. 2018b. SFV: Reinforcement Learning of Physical Skills from Videos. ACM Transactions on Graphics (TOG) (2018).","DOI":"10.1145\/3272127.3275014"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"crossref","unstructured":"Xue\u00a0Bin Peng Ze Ma Pieter Abbeel Sergey Levine and Angjoo Kanazawa. 2021a. AMP: Adversarial Motion Priors for Stylized Physics-Based Character Control. ACM Transactions on Graphics (TOG) (2021).","DOI":"10.1145\/3450626.3459670"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00276"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01129"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687565"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"crossref","unstructured":"Soshi Shimada Vladislav Golyanik Weipeng Xu and Christian Theobalt. 2020. PhysCap: Physically Plausible Monocular 3D Motion Capture in Real Time. ACM Transactions on Graphics (TOG) (2020).","DOI":"10.1145\/3414685.3417877"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00202"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00855"},{"key":"e_1_3_3_2_37_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Teed Zachary","year":"2023","unstructured":"Zachary Teed, Lahav Lipson, and Jia Deng. 2023. Deep Patch Visual Odometry. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"crossref","unstructured":"Chen Tessler Yunrong Guo Ofir Nabati Gal Chechik and Xue\u00a0Bin Peng. 2024. MaskedMimic: Unified Physics-Based Character Control Through Masked Motion Inpainting. ACM Transactions on Graphics (TOG) (2024).","DOI":"10.1145\/3687951"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"crossref","unstructured":"Chen Tessler Yoni Kasten Yunrong Guo Shie Mannor Gal Chechik and Xue\u00a0Bin Peng. 2023. CALM: Conditional Adversarial Latent Models for Directable Virtual Characters. ACM Transactions on Graphics (TOG) (2023).","DOI":"10.1145\/3588432.3591541"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00457"},{"key":"e_1_3_3_2_42_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Wagener Nolan","year":"2022","unstructured":"Nolan Wagener, Andrey Kolobov, Felipe\u00a0Vieira Frujeri, Ricky Loynd, Ching-An Cheng, and Matthew Hausknecht. 2022. MoCapAct: A Multi-Task Dataset for Simulated Humanoid Control. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_43_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV)","author":"Wang Yufu","year":"2024","unstructured":"Yufu Wang, Ziyun Wang, Lingjie Liu, and Kostas Daniilidis. 2024a. TRAM: Global Trajectory and Motion of 3D Humans from in-the-wild Videos. In Proceedings of the European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_3_2_44_1","unstructured":"Yinhuai Wang Qihan Zhao Runyi Yu Ailing Zeng Jing Lin Zhengyi Luo Hok\u00a0Wai Tsui Jiwen Yu Xiu Li Qifeng Chen Jian Zhang Lei Zhang and Ping Tan. 2024b. SkillMimic: Learning Reusable Basketball Skills from Demonstrations. arxiv:https:\/\/arXiv.org\/abs\/2408.15270v1\u00a0[cs.CV]"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550469.3555411"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550469.3555411"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01122"},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00622"},{"key":"e_1_3_3_2_49_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","author":"Xu Yufei","year":"2022","unstructured":"Yufei Xu, Jing Zhang, Qiming Zhang, and Dacheng Tao. 2022. ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00362"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02033"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"crossref","unstructured":"Wanqi Yin Zhongang Cai Ruisi Wang Ailing Zeng Chen Wei Qingping Sun Haiyi Mei Yanjun Wang Hui\u00a0En Pang Mingyuan Zhang Lei Zhang Chen\u00a0Change Loy Atsushi Yamashita Lei Yang and Ziwei Liu. 2025. SMPLest-X: Ultimate Scaling for Expressive Human Pose and Shape Estimation. arxiv:https:\/\/arXiv.org\/abs\/2501.09782\u00a0[cs.CV]","DOI":"10.1109\/TPAMI.2025.3618174"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01076"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00708"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00224"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00605"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763951","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:25:27Z","timestamp":1765250727000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763951"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":55,"alternative-id":["10.1145\/3757377.3763951","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763951","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}