{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:14:46Z","timestamp":1782317686054,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":73,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,11,29]],"date-time":"2022-11-29T00:00:00Z","timestamp":1669680000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,11,29]]},"DOI":"10.1145\/3550469.3555423","type":"proceedings-article","created":{"date-parts":[[2022,11,30]],"date-time":"2022-11-30T11:07:54Z","timestamp":1669806474000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":84,"title":["Capturing and Animation of Body and Clothing from Monocular Video"],"prefix":"10.1145","author":[{"given":"Yao","family":"Feng","sequence":"first","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany and ETH Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinlong","family":"Yang","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Marc","family":"Pollefeys","sequence":"additional","affiliation":[{"name":"ETH Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael J.","family":"Black","sequence":"additional","affiliation":[{"name":"Max Planck Institute for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Timo","family":"Bolkart","sequence":"additional","affiliation":[{"name":"Max Planck Institue for Intelligent Systems, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,11,30]]},"reference":[{"key":"e_1_3_2_3_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00127"},{"key":"e_1_3_2_3_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2018.00022"},{"key":"e_1_3_2_3_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00875"},{"key":"e_1_3_2_3_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00238"},{"key":"e_1_3_2_3_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00541"},{"key":"e_1_3_2_3_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1186822.1073207"},{"key":"e_1_3_2_3_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58565-5_21"},{"key":"e_1_3_2_3_8_1","unstructured":"Jianchuan Chen Ying Zhang Di Kang Xuefei Zhe Linchao Bao Xu Jia and Huchuan Lu. 2021b. Animatable Neural Radiance Fields from Monocular RGB Videos. arxiv:2106.13629\u00a0[cs.CV]"},{"key":"e_1_3_2_3_9_1","first-page":"1","article-title":"TightCap: 3D Human Shape Capture with Clothing Tightness Field","volume":"41","author":"Chen Xin","year":"2021","unstructured":"Xin Chen, Anqi Pang, Wei Yang, Peihao Wang, Lan Xu, and Jingyi Yu. 2021a. TightCap: 3D Human Shape Capture with Clothing Tightness Field. Transactions on Graphics (TOG) 41, 1 (2021), 1\u201317.","journal-title":"Transactions on Graphics (TOG)"},{"key":"e_1_3_2_3_10_1","unstructured":"Julian Chibane Aymen Mir and Gerard Pons-Moll. 2020. Neural Unsigned Distance Fields for Implicit Function Learning. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_3_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58607-2_2"},{"key":"e_1_3_2_3_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01170"},{"key":"e_1_3_2_3_13_1","unstructured":"Levin Dabhi. 2022. Clothes Segmentation using U2NET. https:\/\/github.com\/levindabhi\/cloth-segmentation"},{"key":"e_1_3_2_3_14_1","volume-title":"EMOCA: Emotion Driven Monocular Face Capture and Animation. In Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Danecek Radek","year":"2022","unstructured":"Radek Danecek, Michael\u00a0J. Black, and Timo Bolkart. 2022. EMOCA: Emotion Driven Monocular Face Capture and Animation. In Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_3_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00088"},{"key":"e_1_3_2_3_16_1","article-title":"Learning an Animatable Detailed 3D Face Model from In-the-Wild Images","volume":"40","author":"Feng Yao","year":"2021","unstructured":"Yao Feng, Haiwen Feng, Michael\u00a0J. Black, and Timo Bolkart. 2021b. Learning an Animatable Detailed 3D Face Model from In-the-Wild Images. Transactions on Graphics, (Proc. SIGGRAPH) 40, 4 (2021), 88:1\u201388:13.","journal-title":"Transactions on Graphics, (Proc. SIGGRAPH)"},{"key":"e_1_3_2_3_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01810"},{"key":"e_1_3_2_3_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33783-3_18"},{"key":"e_1_3_2_3_19_1","volume-title":"Conference on Computer Vision and Pattern Recognition (CVPR). 20374\u201320384","author":"Hong Yang","year":"2022","unstructured":"Yang Hong, Bo Peng, Haiyao Xiao, Ligang Liu, and Juyong Zhang. 2022. HeadNeRF: A real-time nerf-based parametric head model. In Conference on Computer Vision and Pattern Recognition (CVPR). 20374\u201320384."},{"key":"e_1_3_2_3_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00316"},{"key":"e_1_3_2_3_21_1","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177703732"},{"key":"e_1_3_2_3_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00552"},{"key":"e_1_3_2_3_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58565-5_2"},{"key":"e_1_3_2_3_24_1","volume-title":"Computer Graphics Forum, Vol.\u00a039","author":"Jin Ning","unstructured":"Ning Jin, Yilin Zhu, Zhenglin Geng, and Ronald Fedkiw. 2020. A Pixel-Based Framework for Data-Driven Clothing. In Computer Graphics Forum, Vol.\u00a039. Wiley Online Library, 135\u2013144."},{"key":"e_1_3_2_3_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00868"},{"key":"e_1_3_2_3_26_1","volume-title":"End-to-end Recovery of Human Shape and Pose. In Conference on Computer Vision and Pattern Recognition (CVPR). 7122\u20137131","author":"Kanazawa Angjoo","year":"2018","unstructured":"Angjoo Kanazawa, Michael\u00a0J. Black, David\u00a0W. Jacobs, and Jitendra Malik. 2018. End-to-end Recovery of Human Shape and Pose. In Conference on Computer Vision and Pattern Recognition (CVPR). 7122\u20137131."},{"key":"e_1_3_2_3_27_1","volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations (ICLR).","author":"P.","unstructured":"Diederik\u00a0P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_3_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00234"},{"key":"e_1_3_2_3_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2019.00076"},{"key":"e_1_3_2_3_30_1","unstructured":"Shanchuan Lin Linjie Yang Imran Saleemi and Soumyadip Sengupta. 2022. Robust Video Matting (RVM). https:\/\/github.com\/PeterL1n\/RobustVideoMatting"},{"key":"e_1_3_2_3_31_1","first-page":"1","article-title":"Neural actor: Neural free-view synthesis of human actors with pose control","volume":"40","author":"Liu Lingjie","year":"2021","unstructured":"Lingjie Liu, Marc Habermann, Viktor Rudnev, Kripasindhu Sarkar, Jiatao Gu, and Christian Theobalt. 2021b. Neural actor: Neural free-view synthesis of human actors with pose control. Transactions on Graphics (TOG) 40, 6 (2021), 1\u201316.","journal-title":"Transactions on Graphics (TOG)"},{"key":"e_1_3_2_3_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00209"},{"key":"e_1_3_2_3_33_1","doi-asserted-by":"crossref","unstructured":"Wu Liu Qian Bao Yu Sun and Tao Mei. 2021a. Recent Advances in Monocular 2D and 3D Human Pose Estimation: A Deep Learning Perspective. CoRR abs\/2104.11536(2021).","DOI":"10.1145\/3524497"},{"key":"e_1_3_2_3_34_1","volume-title":"Appearance Transfer and Novel View Synthesis. In International Conference on Computer Vision (ICCV). 5903\u20135912","author":"Liu Wen","year":"2019","unstructured":"Wen Liu, Zhixin Piao, Min Jie, Wenhan Luo, Lin Ma, and Shenghua Gao. 2019. Liquid Warping GAN: A Unified Framework for Human Motion Imitation, Appearance Transfer and Novel View Synthesis. In International Conference on Computer Vision (ICCV). 5903\u20135912."},{"key":"e_1_3_2_3_35_1","article-title":"SMPL: A Skinned Multi-Person Linear Model","volume":"34","author":"Loper Matthew","year":"2015","unstructured":"Matthew Loper, Naureen Mahmood, Javier Romero, Gerard Pons-Moll, and Michael\u00a0J. Black. 2015. SMPL: A Skinned Multi-Person Linear Model. Transactions on Graphics, (Proc. SIGGRAPH Asia) 34, 6 (2015), 248:1\u2013248:16.","journal-title":"Transactions on Graphics, (Proc. SIGGRAPH Asia)"},{"key":"e_1_3_2_3_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00650"},{"key":"e_1_3_2_3_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00650"},{"key":"e_1_3_2_3_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01032"},{"key":"e_1_3_2_3_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_24"},{"key":"e_1_3_2_3_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00356"},{"key":"e_1_3_2_3_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2018.00062"},{"key":"e_1_3_2_3_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58539-6_36"},{"key":"e_1_3_2_3_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00739"},{"key":"e_1_3_2_3_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01123"},{"key":"e_1_3_2_3_45_1","doi-asserted-by":"crossref","unstructured":"Sida Peng Junting Dong Qianqian Wang Shangzhan Zhang Qing Shuai Xiaowei Zhou and Hujun Bao. 2021a. Animatable Neural Radiance Fields for Modeling Dynamic Human Bodies. In ICCV.","DOI":"10.1109\/ICCV48922.2021.01405"},{"key":"e_1_3_2_3_46_1","unstructured":"Sida Peng Shangzhan Zhang Zhen Xu Chen Geng Boyi Jiang Hujun Bao and Xiaowei Zhou. 2022. Animatable Neural Implicit Surfaces for Creating Avatars from Videos. arXiv preprint arXiv:2203.08133(2022)."},{"key":"e_1_3_2_3_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00894"},{"key":"e_1_3_2_3_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073711"},{"key":"e_1_3_2_3_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00185"},{"key":"e_1_3_2_3_50_1","volume-title":"Accelerating 3D Deep Learning with PyTorch3D. arXiv:2007.08501","author":"Ravi Nikhila","year":"2020","unstructured":"Nikhila Ravi, Jeremy Reizenstein, David Novotny, Taylor Gordon, Wan-Yen Lo, Justin Johnson, and Georgia Gkioxari. 2020. Accelerating 3D Deep Learning with PyTorch3D. arXiv:2007.08501 (2020)."},{"key":"e_1_3_2_3_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00201"},{"key":"e_1_3_2_3_52_1","volume-title":"PIFu: Pixel-Aligned Implicit Function for High-Resolution Clothed Human Digitization. In International Conference on Computer Vision (ICCV).","author":"Saito Shunsuke","year":"2019","unstructured":"Shunsuke Saito, Zeng Huang, Ryota Natsume, Shigeo Morishima, Angjoo Kanazawa, and Hao Li. 2019. PIFu: Pixel-Aligned Implicit Function for High-Resolution Clothed Human Digitization. In International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_3_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00016"},{"key":"e_1_3_2_3_54_1","volume-title":"Computer Graphics Forum, Vol.\u00a038","author":"Santesteban Igor","unstructured":"Igor Santesteban, Miguel\u00a0A Otaduy, and Dan Casas. 2019. Learning-based animation of clothing for virtual try-on. In Computer Graphics Forum, Vol.\u00a038. Wiley Online Library, 355\u2013366."},{"key":"e_1_3_2_3_55_1","volume-title":"A-NeRF: Articulated neural radiance fields for learning human shape, appearance, and pose. Advances in Neural Information Processing Systems (NeurIPS) 34","author":"Su Shih-Yang","year":"2021","unstructured":"Shih-Yang Su, Frank Yu, Michael Zollh\u00f6fer, and Helge Rhodin. 2021. A-NeRF: Articulated neural radiance fields for learning human shape, appearance, and pose. Advances in Neural Information Processing Systems (NeurIPS) 34 (2021)."},{"key":"e_1_3_2_3_56_1","unstructured":"Yating Tian Hongwen Zhang Yebin Liu and Limin Wang. 2022. Recovering 3D Human Mesh from Monocular Images: A Survey. arXiv preprint arXiv:2203.01923(2022)."},{"key":"e_1_3_2_3_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_1"},{"key":"e_1_3_2_3_58_1","volume-title":"Computer Graphics Forum, Vol.\u00a039","author":"Vidaurre Raquel","unstructured":"Raquel Vidaurre, Igor Santesteban, Elena Garces, and Dan Casas. 2020. Fully Convolutional Graph Neural Networks for Parametric Virtual Try-On. In Computer Graphics Forum, Vol.\u00a039. Wiley Online Library, 145\u2013156."},{"key":"e_1_3_2_3_59_1","unstructured":"Yi Wang Xin Tao Xiaojuan Qi Xiaoyong Shen and Jiaya Jia. 2018. Image inpainting via generative multi-column convolutional neural networks. In Advances in Neural Information Processing Systems (NeurIPS). 331\u2013340."},{"key":"e_1_3_2_3_60_1","volume-title":"HumanNeRF: Free-Viewpoint Rendering of Moving People From Monocular Video. In Conference on Computer Vision and Pattern Recognition (CVPR). 16210\u201316220","author":"Weng Chung-Yi","year":"2022","unstructured":"Chung-Yi Weng, Brian Curless, Pratul\u00a0P. Srinivasan, Jonathan\u00a0T. Barron, and Ira Kemelmacher-Shlizerman. 2022. HumanNeRF: Free-Viewpoint Rendering of Moving People From Monocular Video. In Conference on Computer Vision and Pattern Recognition (CVPR). 16210\u201316220."},{"key":"e_1_3_2_3_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01122"},{"key":"e_1_3_2_3_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459882"},{"key":"e_1_3_2_3_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01294"},{"key":"e_1_3_2_3_64_1","volume-title":"H-NeRF: Neural radiance fields for rendering and temporal reconstruction of humans in motion. Advances in Neural Information Processing Systems (NeurIPS) 34","author":"Xu Hongyi","year":"2021","unstructured":"Hongyi Xu, Thiemo Alldieck, and Cristian Sminchisescu. 2021. H-NeRF: Neural radiance fields for rendering and temporal reconstruction of humans in motion. Advances in Neural Information Processing Systems (NeurIPS) 34 (2021)."},{"key":"e_1_3_2_3_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00622"},{"key":"e_1_3_2_3_66_1","volume-title":"Renovating Parsing R-CNN for Accurate Multiple Human Parsing. In European Conference on Computer Vision (ECCV)(Lecture Notes in Computer Science, Vol.\u00a012357)","author":"Yang Lu","year":"2020","unstructured":"Lu Yang, Qing Song, Zhihui Wang, Mengjie Hu, Chun Liu, Xueshi Xin, Wenhe Jia, and Songcen Xu. 2020. Renovating Parsing R-CNN for Accurate Multiple Human Parsing. In European Conference on Computer Vision (ECCV)(Lecture Notes in Computer Science, Vol.\u00a012357). Springer, 421\u2013437."},{"key":"e_1_3_2_3_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01308"},{"key":"e_1_3_2_3_68_1","unstructured":"Lior Yariv Jiatao Gu Yoni Kasten and Yaron Lipman. 2021. Volume rendering of neural implicit surfaces. Advances in Neural Information Processing Systems (NeurIPS) 34 (2021)."},{"key":"e_1_3_2_3_69_1","first-page":"2492","article-title":"Multiview neural surface reconstruction by disentangling geometry and appearance","volume":"33","author":"Yariv Lior","year":"2020","unstructured":"Lior Yariv, Yoni Kasten, Dror Moran, Meirav Galun, Matan Atzmon, Basri Ronen, and Yaron Lipman. 2020. Multiview neural surface reconstruction by disentangling geometry and appearance. Advances in Neural Information Processing Systems (NeurIPS) 33 (2020), 2492\u20132502.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_3_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01425"},{"key":"e_1_3_2_3_71_1","volume-title":"PaMIR: Parametric model-conditioned implicit representation for image-based human reconstruction. Transactions on Pattern Analysis and Machine Intelligence (PAMI)","author":"Zheng Zerong","year":"2021","unstructured":"Zerong Zheng, Tao Yu, Yebin Liu, and Qionghai Dai. 2021. PaMIR: Parametric model-conditioned implicit representation for image-based human reconstruction. Transactions on Pattern Analysis and Machine Intelligence (PAMI) (2021)."},{"key":"e_1_3_2_3_72_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_30"},{"key":"e_1_3_2_3_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00382"}],"event":{"name":"SA '22: SIGGRAPH Asia 2022","location":"Daegu Republic of Korea","acronym":"SA '22","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["SIGGRAPH Asia 2022 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3550469.3555423","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3550469.3555423","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:51:44Z","timestamp":1750182704000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3550469.3555423"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,29]]},"references-count":73,"alternative-id":["10.1145\/3550469.3555423","10.1145\/3550469"],"URL":"https:\/\/doi.org\/10.1145\/3550469.3555423","relation":{},"subject":[],"published":{"date-parts":[[2022,11,29]]},"assertion":[{"value":"2022-11-30","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}