{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:06:35Z","timestamp":1784228795213,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Nature Science Foundation of China (NSFC)","award":["62225603"],"award-info":[{"award-number":["62225603"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811135","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["R-DMesh: Video-Guided 3D Animation via Rectified Dynamic Mesh Flow"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-1171-9763","authenticated-orcid":false,"given":"Zijie","family":"Wu","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China and Tencent Hunyuan, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3579-3052","authenticated-orcid":false,"given":"Lixin","family":"Xu","sequence":"additional","affiliation":[{"name":"Tencent Hunyuan, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1662-0736","authenticated-orcid":false,"given":"Puhua","family":"Jiang","sequence":"additional","affiliation":[{"name":"Tencent Hunyuan, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3652-1836","authenticated-orcid":false,"given":"Sicong","family":"Liu","sequence":"additional","affiliation":[{"name":"Tencent Hunyuan, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7465-802X","authenticated-orcid":false,"given":"Chunchao","family":"Guo","sequence":"additional","affiliation":[{"name":"Tencent Hunyuan, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3449-5940","authenticated-orcid":false,"given":"Xiang","family":"Bai","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_3_2_1","doi-asserted-by":"crossref","unstructured":"Sherwin Bahmani Ivan Skorokhodov Victor Rong Gordon Wetzstein Leonidas Guibas Peter Wonka Sergey Tulyakov Jeong\u00a0Joon Park Andrea Tagliasacchi and David\u00a0B Lindell. 2023. 4d-fy: Text-to-4d generation using hybrid score distillation sampling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.17984 (2023).","DOI":"10.1109\/CVPR52733.2024.00764"},{"key":"e_1_3_3_3_3_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et\u00a0al. 2025. Qwen2. 5-VL Technical Report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_3_4_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et\u00a0al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.15127 (2023)."},{"key":"e_1_3_3_3_5_1","unstructured":"Nicolas Carion Laura Gustafson Yuan-Ting Hu Shoubhik Debnath Ronghang Hu Didac Suris Chaitanya Ryali Kalyan\u00a0Vasudev Alwala Haitham Khedr Andrew Huang et\u00a0al. 2025. Sam 3: Segment anything with concepts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.16719 (2025)."},{"key":"e_1_3_3_3_6_1","unstructured":"Cerspense. 2023. Zeroscope text-to-video model. https:\/\/huggingface.co\/cerspense\/zeroscope_v2_576w. Accessed: 2023-10-31."},{"key":"e_1_3_3_3_7_1","unstructured":"Ce Chen Shaoli Huang Xuelin Chen Guangyi Chen Xiaoguang Han Kun Zhang and Mingming Gong. 2024. Ct4d: Consistent text-to-4d generation with animatable meshes. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.08342 (2024)."},{"key":"e_1_3_3_3_8_1","doi-asserted-by":"crossref","unstructured":"Jianqi Chen Biao Zhang Xiangjun Tang and Peter Wonka. 2025. V2M4: 4D Mesh Animation Reconstruction from a Single Monocular Video. ArXiv abs\/2503.09631 (2025).","DOI":"10.1109\/ICCV51701.2025.01083"},{"key":"e_1_3_3_3_9_1","doi-asserted-by":"crossref","unstructured":"Matt Deitke Ruoshi Liu Matthew Wallingford Huong Ngo Oscar Michel Aditya Kusupati Alan Fan Christian Laforte Vikram Voleti Samir\u00a0Yitzhak Gadre et\u00a0al. 2024. Objaverse-xl: A universe of 10m+ 3d objects. Advances in Neural Information Processing Systems 36 (2024).","DOI":"10.52202\/075280-1554"},{"key":"e_1_3_3_3_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"e_1_3_3_3_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730621"},{"key":"e_1_3_3_3_12_1","unstructured":"Kehong Gong Zhengyu Wen Weixia He Mingxi Xu Qi Wang Ning Zhang Zhengyu Li Dongze Lian Wei Zhao Xiaoyu He et\u00a0al. 2025. MoCapAnything: Unified 3D Motion Capture for Arbitrary Skeletons from Monocular Videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2512.10881 (2025)."},{"key":"e_1_3_3_3_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00186"},{"key":"e_1_3_3_3_14_1","unstructured":"Jonathan Ho and Tim Salimans. 2022. Classifier-free diffusion guidance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2207.12598 (2022)."},{"key":"e_1_3_3_3_15_1","unstructured":"Yicong Hong Kai Zhang Jiuxiang Gu Sai Bi Yang Zhou Difan Liu Feng Liu Kalyan Sunkavalli Trung Bui and Hao Tan. 2023. Lrm: Large reconstruction model for single image to 3d. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.04400 (2023)."},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02060"},{"key":"e_1_3_3_3_17_1","doi-asserted-by":"crossref","unstructured":"Yanqin Jiang Chaohui Yu Chenjie Cao Fan Wang Weiming Hu and Jin Gao. 2024. Animate3d: Animating any 3d model with multi-view video diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.11398 (2024).","DOI":"10.52202\/079017-3999"},{"key":"e_1_3_3_3_18_1","unstructured":"Yanqin Jiang Li Zhang Jin Gao Weimin Hu and Yao Yao. 2023. Consistent4d: Consistent 360 {\\ deg} dynamic object generation from monocular video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.02848 (2023)."},{"key":"e_1_3_3_3_19_1","unstructured":"Zeren Jiang Chuanxia Zheng Iro Laina Diane Larlus and Andrea Vedaldi. 2026. Mesh4D: 4D Mesh Reconstruction and Tracking from Monocular Video. ArXiv abs\/2601.05251 (2026)."},{"key":"e_1_3_3_3_20_1","unstructured":"Zeqiang Lai Yunfei Zhao Haolin Liu Zibo Zhao Qingxiang Lin Huiwen Shi Xianghui Yang Mingxin Yang Shuhui Yang Yifei Feng et\u00a0al. 2025. Hunyuan3D 2.5: Towards High-Fidelity 3D Assets Generation with Ultimate Details. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.16504 (2025)."},{"key":"e_1_3_3_3_21_1","doi-asserted-by":"crossref","unstructured":"Hanwen Liang Yuyang Yin Dejia Xu Hanxue Liang Zhangyang Wang Konstantinos\u00a0N Plataniotis Yao Zhao and Yunchao Wei. 2024. Diffusion4d: Fast spatial-temporal consistent 4d generation via video diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.16645 (2024).","DOI":"10.52202\/079017-3519"},{"key":"e_1_3_3_3_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"e_1_3_3_3_23_1","unstructured":"Xingchao Liu Chengyue Gong and Qiang Liu. 2022. Flow straight and fast: Learning to generate and transfer data with rectified flow. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.03003 (2022)."},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"crossref","unstructured":"Matthew Loper Naureen Mahmood Javier Romero Gerard Pons-Moll and Michael\u00a0J Black. 2023. SMPL: A skinned multi-person linear model. Seminal Graphics Papers: Pushing the Boundaries Volume 2 (2023) 851\u2013866.","DOI":"10.1145\/3596711.3596800"},{"key":"e_1_3_3_3_25_1","unstructured":"Marc Bened\u00ed\u00a0San Mill\u00e1n Angela Dai and Matthias Nie\u00dfner. 2025. Animating the Uncaptured: Humanoid Mesh Animation with Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.15996 (2025)."},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_3_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"e_1_3_3_3_28_1","unstructured":"Ben Poole Ajay Jain Jonathan\u00a0T Barron and Ben Mildenhall. 2022. Dreamfusion: Text-to-3d using 2d diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14988 (2022)."},{"key":"e_1_3_3_3_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01018"},{"key":"e_1_3_3_3_30_1","unstructured":"Charles\u00a0Ruizhongtai Qi Li Yi Hao Su and Leonidas\u00a0J Guibas. 2017. Pointnet++: Deep hierarchical feature learning on point sets in a metric space. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"crossref","unstructured":"Jiawei Ren Cheng Xie Ashkan Mirzaei Karsten Kreis Ziwei Liu Antonio Torralba Sanja Fidler Seung\u00a0Wook Kim Huan Ling et\u00a0al. 2025. L4gm: Large 4d gaussian reconstruction model. Advances in Neural Information Processing Systems 37 (2025) 56828\u201356858.","DOI":"10.52202\/079017-1810"},{"key":"e_1_3_3_3_32_1","unstructured":"Ruoxi Shi Hansheng Chen Zhuoyang Zhang Minghua Liu Chao Xu Xinyue Wei Linghao Chen Chong Zeng and Hao Su. 2023. Zero123++: a single image to consistent multi-view diffusion base model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.15110 (2023)."},{"key":"e_1_3_3_3_33_1","unstructured":"Yahao Shi Yang Liu Yanmin Wu Xing Liu Chen Zhao Jie Luo and Bin Zhou. 2025. Drive Any Mesh: 4D Latent Diffusion for Mesh Deformation from Video. ArXiv abs\/2506.07489 (2025)."},{"key":"e_1_3_3_3_34_1","unstructured":"Uriel Singer Shelly Sheynin Adam Polyak Oron Ashual Iurii Makarov Filippos Kokkinos Naman Goyal Andrea Vedaldi Devi Parikh Justin Johnson et\u00a0al. 2023. Text-to-4d dynamic scene generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.11280 (2023)."},{"key":"e_1_3_3_3_35_1","unstructured":"Chaoyue Song Xiu Li Fan Yang Zhongcong Xu Jiacheng Wei Fayao Liu Jiashi Feng Guosheng Lin and Jianfeng Zhang. 2025. Puppeteer: Rig and animate your 3d models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.10898 (2025)."},{"key":"e_1_3_3_3_36_1","first-page":"1","volume-title":"European Conference on Computer Vision","author":"Tang Jiaxiang","year":"2024","unstructured":"Jiaxiang Tang, Zhaoxi Chen, Xiaokang Chen, Tengfei Wang, Gang Zeng, and Ziwei Liu. 2024. Lgm: Large multi-view gaussian model for high-resolution 3d content creation. In European Conference on Computer Vision. Springer, 1\u201318."},{"key":"e_1_3_3_3_37_1","unstructured":"Gusi Te Xiu Li Xiao Li Jinglu Wang Wei Hu and Yan Lu. 2022. Neural Capture of Animatable 3D Human from Monocular Video. ArXiv abs\/2208.08728 (2022)."},{"key":"e_1_3_3_3_38_1","unstructured":"Dmitry Tochilkin David Pankratz Zexiang Liu Zixuan Huang Adam Letts Yangguang Li Ding Liang Christian Laforte Varun Jampani and Yan-Pei Cao. 2024. Triposr: Fast 3d object reconstruction from a single image. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.02151 (2024)."},{"key":"e_1_3_3_3_39_1","unstructured":"Lukas Uzolas Elmar Eisemann and Petr Kellnhofer. 2024. Motiondreamer: Zero-shot 3d mesh animation from video diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.20155 (2024)."},{"key":"e_1_3_3_3_40_1","unstructured":"Team Wan Ang Wang Baole Ai Bin Wen Chaojie Mao Chen-Wei Xie Di Chen Feiwu Yu Haiming Zhao Jianxiao Yang et\u00a0al. 2025. Wan: Open and advanced large-scale video generative models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.20314 (2025)."},{"key":"e_1_3_3_3_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01920"},{"key":"e_1_3_3_3_42_1","first-page":"361","volume-title":"European Conference on Computer Vision","author":"Wu Zijie","year":"2024","unstructured":"Zijie Wu, Chaohui Yu, Yanqin Jiang, Chenjie Cao, Fan Wang, and Xiang Bai. 2024b. Sc4d: Sparse-controlled video-to-4d generation and motion transfer. In European Conference on Computer Vision. Springer, 361\u2013379."},{"key":"e_1_3_3_3_43_1","unstructured":"Zijie Wu Chaohui Yu Fan Wang and Xiang Bai. 2025. AnimateAnyMesh: A Feed-Forward 4D Foundation Model for Text-Driven Universal Mesh Animation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.09982 (2025)."},{"key":"e_1_3_3_3_44_1","unstructured":"Zijie Wu Chaohui Yu Fan Wang and Xiang Bai. 2026. AnimateAnyMesh++: A Flexible 4D Foundation Model for High-Fidelity Text-Driven Mesh Animation. arxiv:https:\/\/arXiv.org\/abs\/2604.26917\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2604.26917"},{"key":"e_1_3_3_3_45_1","unstructured":"Zhuoyi Yang Jiayan Teng Wendi Zheng Ming Ding Shiyu Huang Jiazheng Xu Yuanming Yang Wenyi Hong Xiaohan Zhang Guanyu Feng et\u00a0al. 2024. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.06072 (2024)."},{"key":"e_1_3_3_3_46_1","doi-asserted-by":"crossref","unstructured":"Biao Zhang Jiapeng Tang Matthias Niessner and Peter Wonka. 2023a. 3dshape2vecset: A 3d shape representation for neural fields and generative diffusion models. ACM Transactions On Graphics (TOG) 42 4 (2023) 1\u201316.","DOI":"10.1145\/3592442"},{"key":"e_1_3_3_3_47_1","doi-asserted-by":"crossref","unstructured":"Haiyu Zhang Xinyuan Chen Yaohui Wang Xihui Liu Yunhong Wang and Yu Qiao. 2025. 4diffusion: Multi-view video diffusion model for 4d generation. Advances in Neural Information Processing Systems 37 (2025) 15272\u201315295.","DOI":"10.52202\/079017-0488"},{"key":"e_1_3_3_3_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"e_1_3_3_3_49_1","doi-asserted-by":"crossref","unstructured":"Longwen Zhang Ziyu Wang Qixuan Zhang Qiwei Qiu Anqi Pang Haoran Jiang Wei Yang Lan Xu and Jingyi Yu. 2024b. Clay: A controllable large-scale generative model for creating high-quality 3d assets. ACM Transactions on Graphics (TOG) 43 4 (2024) 1\u201320.","DOI":"10.1145\/3658146"},{"key":"e_1_3_3_3_50_1","doi-asserted-by":"crossref","unstructured":"Mingyuan Zhang Zhongang Cai Liang Pan Fangzhou Hong Xinying Guo Lei Yang and Ziwei Liu. 2024a. Motiondiffuse: Text-driven human motion generation with diffusion model. IEEE transactions on pattern analysis and machine intelligence 46 6 (2024) 4115\u20134128.","DOI":"10.1109\/TPAMI.2024.3355414"}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:27:14Z","timestamp":1784226434000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811135"}},"subtitle":["R-DMesh"],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":49,"alternative-id":["10.1145\/3799902.3811135","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811135","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}