{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T16:25:24Z","timestamp":1780331124603,"version":"3.54.1"},"reference-count":93,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00513","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"5290-5301","source":"Crossref","is-referenced-by-count":1,"title":["FB-4D: Spatial-Temporal Coherent Dynamic 3D Content Generation with Feature Banks"],"prefix":"10.1109","author":[{"given":"Jinwei","family":"Li","sequence":"first","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huan-Ang","family":"Gao","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenyi","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haohan","family":"Chi","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenxi","family":"Du","sequence":"additional","affiliation":[{"name":"Tsinghua University,DCST"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiqian","family":"Liu","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingju","family":"Gao","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guiyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zongzheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Yi","sequence":"additional","affiliation":[{"name":"Tsinghua University,IIIS"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yao","family":"Yao","sequence":"additional","affiliation":[{"name":"Nanjing University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingwei","family":"Zhao","sequence":"additional","affiliation":[{"name":"Xiaomi Corporation"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyang","family":"Li","sequence":"additional","affiliation":[{"name":"The University of Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yikai","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing Normal University,School of Artificial Intelligence"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Zhao","sequence":"additional","affiliation":[{"name":"Tsinghua University,AIR"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72952-2_4"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00764"},{"key":"ref3","article-title":"Vd3d: Taming large video diffusion transformers for 3d camera control","author":"Bahmani","year":"2024"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73036-8_15"},{"key":"ref5","article-title":"Stable video diffusion: Scaling latent video diffusion models to large datasets","author":"Blattmann","year":"2023"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"ref7","article-title":"Token merging: Your vit but faster","author":"Bolya","year":"2022"},{"key":"ref8","article-title":"Ultraman: single image 3d human reconstruction with ultra speed and detail","author":"Chen","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02022"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3048"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICME59968.2025.11209788"},{"key":"ref12","first-page":"37","article-title":"Scp-diff: Spatial-categorical joint prior for diffusion based semantic image synthesis","volume-title":"European Conference on Computer Vision","author":"Gao"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00512"},{"key":"ref14","article-title":"Stable-dreamer: Taming noisy score distillation sampling for text-to-3d","author":"Guo","year":"2023"},{"key":"ref15","article-title":"Animatediff: Animate your personalized text-to-image diffusion models without specific tuning","author":"Guo","year":"2023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01976"},{"key":"ref17","article-title":"Cameractrl: Enabling camera control for text-to-video generation","author":"He","year":"2024"},{"key":"ref18","article-title":"Latent video diffusion models for high-fidelity long video generation","author":"He","year":"2022"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0628"},{"key":"ref20","article-title":"Lrm: Large reconstruction model for single image to 3d","author":"Hong","year":"2023"},{"key":"ref21","article-title":"Mvtokenflow: High-quality 4d content generation using multiview token flow","author":"Huang","year":"2025"},{"key":"ref22","article-title":"Leap: Liberate sparse-view 3d modeling from camera poses","author":"Jiang","year":"2023"},{"key":"ref23","article-title":"Consistent4d: Consistent 360 {\\deg} dynamic object generation from monocular video","author":"Jiang","year":"2023"},{"key":"ref24","first-page":"125879","article-title":"Animate3d: Animating any 3d model with multi-view video diffusion","volume":"37","author":"Jiang","year":"2025","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02100"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11127968"},{"key":"ref28","article-title":"Instant3d: Fast text-to-3d with sparse-view generation and large reconstruction model","author":"Li","year":"2023"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1780"},{"key":"ref30","article-title":"4k4dgen: Panoramic 4d generation at 4k resolution","author":"Li","year":"2024"},{"key":"ref31","article-title":"Sweet-dreamer: Aligning geometric priors in 2d diffusion for consistent text-to-3d","author":"Li","year":"2023"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_24"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72384-1_58"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00623"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00819"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657402"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00960"},{"key":"ref38","article-title":"One-2-3-45: Any single image to 3d mesh in 45 seconds without per-shape optimization","volume":"36","author":"Liu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"ref40","article-title":"Syncdreamer: Generating multiview-consistent images from a single-view image","author":"Liu","year":"2023"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00951"},{"key":"ref42","article-title":"Diffusion hyperfeatures: Searching through time and space for semantic correspondence","volume":"36","author":"Luo","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3528223.3530127"},{"key":"ref44","article-title":"Straight-line diffusion model for efficient 3d molecular generation","author":"Ni","year":"2025"},{"key":"ref45","article-title":"Zero4d: Training-free 4d video generation from single video using off-the-shelf video diffusion","author":"Park","year":"2025"},{"key":"ref46","article-title":"Dreamgaussian4d: Generative 4d gaussian splatting","author":"Ren","year":"2023"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1810"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.00900"},{"key":"ref50","article-title":"Zero123++: a single image to consistent multi-view diffusion base model","author":"Shi","year":"2023"},{"key":"ref51","article-title":"Toss: High-quality text-guided novel view synthesis from a single image","author":"Shi","year":"2023"},{"key":"ref52","article-title":"Mvdream: Multi-view diffusion for 3d generation","author":"Shi","year":"2023"},{"key":"ref53","article-title":"Drive any mesh: 4d latent diffusion for mesh deformation from video","author":"Shi","year":"2025"},{"key":"ref54","article-title":"Make-a-video: Text-to-video generation without text-video data","author":"Singer","year":"2022"},{"key":"ref55","article-title":"Text-to-4d dynamic scene generation","author":"Singer","year":"2023"},{"key":"ref56","article-title":"Dreamcraft3d: Hierarchical 3d generation with bootstrapped diffusion prior","author":"Sun","year":"2023"},{"key":"ref57","article-title":"Eg4d: Explicit generation of 4d object without score distillation","author":"Sun","year":"2024"},{"key":"ref58","article-title":"Dreamgaussian: Generative gaussian splatting for efficient 3d content creation","author":"Tang","year":"2023"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0068"},{"key":"ref60","article-title":"Triposr: Fast 3d object reconstruction from a single image","author":"Tochilkin","year":"2024"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1698"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73232-4_25"},{"key":"ref63","article-title":"Modelscope text-to-video technical report","author":"Wang","year":"2023"},{"key":"ref64","article-title":"Imagedream: Image-prompt multi-view diffusion for 3d generation","author":"Wang","year":"2023"},{"key":"ref65","article-title":"Pf-lrm: Pose-free large reconstruction model for joint pose and shape prediction","author":"Wang","year":"2023"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2025.104483"},{"key":"ref67","article-title":"Prolificdreamer: High-fidelity and diverse text-to-3d generation with variational score distillation","volume":"36","author":"Wang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657518"},{"key":"ref69","article-title":"Meshlrm: Large reconstruction model for high-quality mesh","author":"Wei","year":"2024"},{"key":"ref70","article-title":"Consistent123: Improve consistency for one image to 3d object synthesis","author":"Weng","year":"2023"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72624-8_21"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72624-8_21"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00420"},{"key":"ref74","article-title":"Sv4d: Dynamic 3d content generation with multi-frame and multi-view consistency","author":"Xie","year":"2024"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00536"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/wacv61041.2025.00099"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1145\/3787849"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00703"},{"key":"ref79","article-title":"Diffusion2: Dynamic 3d content generation via score composition of orthogonal diffusion models","author":"Yang","year":"2024"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01231"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00027"},{"key":"ref82","article-title":"Gaussian-dreamer: Fast generation from text to 3d gaussian splatting with point cloud priors","author":"Yi","year":"2023"},{"key":"ref83","article-title":"4dgen: Grounded 4d content generation with spatial-temporal consistency","author":"Yin","year":"2023"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1438"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01839"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72764-1_10"},{"key":"ref87","article-title":"Ctrl-u: Robust conditional image generation via uncertainty-aware reward modeling","author":"Zhang","year":"2024"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0488"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"ref90","article-title":"Animate124: Animating one image to 4d dynamic scene","author":"Zhao","year":"2023"},{"key":"ref91","article-title":"Genxd: Generating any 3d and 4d scenes","author":"Zhao","year":"2024"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00441"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00983"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492318.pdf?arnumber=11492318","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:56:03Z","timestamp":1778046963000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492318\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":93,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00513","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}