{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T09:25:27Z","timestamp":1780392327073,"version":"3.54.1"},"reference-count":63,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T00:00:00Z","timestamp":1766534400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T00:00:00Z","timestamp":1766534400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s11263-025-02615-z","type":"journal-article","created":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T18:55:12Z","timestamp":1766602512000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["Efficient4D: Fast Dynamic 3D Object Generation from a Single-view Video"],"prefix":"10.1007","volume":"134","author":[{"given":"Zijie","family":"Pan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zeyu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiatian","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1031-5420","authenticated-orcid":false,"given":"Li","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,24]]},"reference":[{"key":"2615_CR1","unstructured":"Bai, J., Xia, M., Fu, X., Wang, X., Mu, L., Cao, J., Liu, Z., Hu, H., Bai, X. Wan, P., and others. (2025). Recammaster: Camera-controlled generative rendering from a single video. arXiv preprint."},{"key":"2615_CR2","unstructured":"Blattmann, A., Dockhorn, T., Kulal, S., Mendelevitch, D., Kilian, M., Lorenz, D., Levi, Y., English, Z., Voleti, V., Letts, A., et al. (2023). Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint."},{"key":"2615_CR3","doi-asserted-by":"crossref","unstructured":"Cao, A., & Johnson, J. (2023). Hexplane: A fast representation for dynamic scenes. In: ICCV.","DOI":"10.1109\/CVPR52729.2023.00021"},{"key":"2615_CR4","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, Y., Jiao, N., & Jia, K. (2023). Fantasia3d: Disentangling geometry and appearance for high-quality text-to-3d content creation. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"2615_CR5","doi-asserted-by":"crossref","unstructured":"Deitke, M., Liu, R., Wallingford, M., Ngo, H., Michel, O., Kusupati, A., Fan, A., Laforte, C., Voleti, V., Gadre, S.Y., and others. (2023). Objaverse-xl: A universe of 10m+ 3d objects. In: NeurIPS.","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"2615_CR6","doi-asserted-by":"crossref","unstructured":"Deitke, M., Schwenk, D., Salvador, J., Weihs, L., Michel, O., VanderBilt, E., Schmidt, L., Ehsani, K., Kembhavi, A., & Farhadi, A. (2023). Objaverse: A universe of annotated 3d objects. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"2615_CR7","doi-asserted-by":"crossref","unstructured":"Esser, P., Chiu, J., Atighehchian, P., Granskog, J., & Germanidis, A. (2023). Structure and content-guided video synthesis with diffusion models. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.00675"},{"key":"2615_CR8","doi-asserted-by":"crossref","unstructured":"Fridovich-Keil, S., Meanti, G., Warburg, F.R., Recht, B., & Kanazawa, A. (2023). K-planes: Explicit radiance fields in space, time, and appearance. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01201"},{"key":"2615_CR9","unstructured":"Geyer, M., Bar-Tal, O., Bagon, S., & Dekel, T. (2023). Tokenflow: Consistent diffusion features for consistent video editing. arXiv preprint."},{"key":"2615_CR10","doi-asserted-by":"crossref","unstructured":"Huang, Y.-H., Sun, Y.-T., Yang, Z., Lyu, X., Cao, Y.-P., & Qi, X. (2024). Sc-gs: Sparse-controlled gaussian splatting for editable dynamic scenes. In: CVPR.","DOI":"10.1109\/CVPR52733.2024.00404"},{"key":"2615_CR11","unstructured":"Huang, Y., Wang, J., Shi, Y., Qi, X., Zha, Z.-J., & Zhang, L. (2023). Dreamtime: An improved optimization strategy for text-to-3d content creation. In: CVPR."},{"key":"2615_CR12","doi-asserted-by":"crossref","unstructured":"Huang, Z., Zhang, T., Heng, W., Shi, B., & Zhou, S. (2022). Real-time intermediate flow estimation for video frame interpolation. In: ECCV.","DOI":"10.1007\/978-3-031-19781-9_36"},{"key":"2615_CR13","unstructured":"Jiang, Y., Zhang, L., Gao, J., Hu, W., & Yao, Y. (2024). Consistent4d: Consistent $$360^{\\circ }$$ dynamic object generation from monocular video. In: ICLR."},{"key":"2615_CR14","unstructured":"Jun, H., & Nichol, A. (2023). Shap-e: Generating conditional 3d implicit functions. arXiv preprint."},{"key":"2615_CR15","doi-asserted-by":"crossref","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., & Drettakis, G. (2023). 3d gaussian splatting for real-time radiance field rendering. In: ACM TOG.","DOI":"10.1145\/3592433"},{"key":"2615_CR16","doi-asserted-by":"crossref","unstructured":"Li, Y., Dou, Y., Shi, Y., Lei, Y., Chen, X., Zhang, Y., Zhou, P., & Ni, B. (2023). Focaldreamer: Text-driven 3d editing via focal-fusion assembly. In: PLMR.","DOI":"10.1609\/aaai.v38i4.28113"},{"key":"2615_CR17","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, Q., Cole, F., Tucker, R., & Snavely, N. (2023). Dynibar: Neural dynamic image-based rendering. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.00416"},{"key":"2615_CR18","doi-asserted-by":"crossref","unstructured":"Lin, C.-H., Gao, J., Tang, L., Takikawa, T., Zeng, X., Huang, X., Kreis, K., Fidler, S., Liu, M.-Y., & Lin, T.-Y. (2023). Magic3d: High-resolution text-to-3d content creation. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.00037"},{"key":"2615_CR19","unstructured":"Liu, J.-W., Cao, Y.-P., Mao, W., Zhang, W., Zhang, D.J., Keppo, J., Shan, Y., Qie, X., & Shou, M.Z. (2022). Devrf: Fast deformable voxel radiance fields for dynamic scenes. In: NeurIPS."},{"key":"2615_CR20","unstructured":"Liu, Y., Lin, C., Zeng, Z., Long, X., Liu, L., Komura, T., & Wang, W. (2024). Syncdreamer: Learning to generate multiview-consistent images from a single-view image. In: ICLR."},{"key":"2615_CR21","doi-asserted-by":"crossref","unstructured":"Liu, R., Wu, R., Van\u00a0Hoorick, B., Tokmakov, P., Zakharov, S., & Vondrick, C. (2023). Zero-1-to-3: Zero-shot one image to 3d object. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"2615_CR22","unstructured":"Liu, M., Xu, C., Jin, H., Chen, L., Xu, Z., Su, H., and others. (2023). One-2-3-45: Any single image to 3d mesh in 45 seconds without per-shape optimization. arXiv preprint."},{"key":"2615_CR23","doi-asserted-by":"crossref","unstructured":"Long, X., Guo, Y.-C., Lin, C., Liu, Y., Dou, Z., Liu, L., Ma, Y., Zhang, S.-H., Habermann, M., Theobalt, C., and others. (2023). Wonder3d: Single image to 3d using cross-domain diffusion. arXiv preprint.","DOI":"10.1109\/CVPR52733.2024.00951"},{"key":"2615_CR24","unstructured":"Maximo. (2023). https:\/\/www.mixamo.com\/"},{"key":"2615_CR25","doi-asserted-by":"crossref","unstructured":"Melas-Kyriazi, L., Laina, I., Rupprecht, C., & Vedaldi, A. (2023). Realfusion: 360deg reconstruction of any object from a single image. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.00816"},{"key":"2615_CR26","doi-asserted-by":"crossref","unstructured":"Metzer, G., Richardson, E., Patashnik, O., Giryes, R., & Cohen-Or, D. (2023). Latent-nerf for shape-guided generation of 3d shapes and textures. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01218"},{"key":"2615_CR27","doi-asserted-by":"crossref","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., & Ng, R. (2021). Nerf: Representing scenes as neural radiance fields for view synthesis. Communications of the ACM.","DOI":"10.1007\/978-3-030-58452-8_24"},{"key":"2615_CR28","unstructured":"Nichol, A., Jun, H., Dhariwal, P., Mishkin, P., & Chen, M. (2022). Point-e: A system for generating 3d point clouds from complex prompts. arXiv preprint."},{"key":"2615_CR29","doi-asserted-by":"crossref","unstructured":"Park, K., Sinha, U., Hedman, P., Barron, J.T., Bouaziz, S., Goldman, D.B., Martin-Brualla, R., & Seitz, S.M. (2021). Hypernerf: a higher-dimensional representation for topologically varying neural radiance fields. In: ACM TOG.","DOI":"10.1145\/3478513.3480487"},{"key":"2615_CR30","unstructured":"Poole, B., Jain, A., Barron, J.T., & Mildenhall, B. (2023). Dreamfusion: Text-to-3d using 2d diffusion. In: ICLR."},{"key":"2615_CR31","doi-asserted-by":"crossref","unstructured":"Pumarola, A., Corona, E., Pons-Moll, G., & Moreno-Noguer, F. (2021). D-nerf: Neural radiance fields for dynamic scenes. In: CVPR.","DOI":"10.1109\/CVPR46437.2021.01018"},{"key":"2615_CR32","unstructured":"Qian, G., Mai, J., Hamdi, A., Ren, J., Siarohin, A., Li, B., Lee, H.-Y., Skorokhodov, I., Wonka, P., Tulyakov, S., and others. (2024). Magic123: One image to high-quality 3d object generation using both 2d and 3d diffusion priors. In: ICLR."},{"key":"2615_CR33","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., and others. (2021). Learning transferable visual models from natural language supervision. In: ICML."},{"key":"2615_CR34","unstructured":"Ren, J., Pan, L., Tang, J., Zhang, C., Cao, A., Zeng, G., & Liu, Z. (2023). Dreamgaussian4d: Generative 4d gaussian splatting. arXiv preprint"},{"key":"2615_CR35","unstructured":"Seo, J., Jang, W., Kwak, M.-S., Ko, J., Kim, H., Kim, J., Kim, J.-H., Lee, J., & Kim, S. (2024). Let 2d diffusion model know 3d-consistency for robust text-to-3d generation. In: ICLR."},{"key":"2615_CR36","doi-asserted-by":"crossref","unstructured":"Shao, R., Zheng, Z., Tu, H., Liu, B., Zhang, H., & Liu, Y. (2023). Tensor4d: Efficient neural 4d decomposition for high-fidelity dynamic reconstruction and rendering. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01596"},{"key":"2615_CR37","unstructured":"Shi, Y., Wang, P., Ye, J., Long, M., Li, K., & Yang, X. (2023). Mvdream: Multi-view diffusion for 3d generation. arXiv preprint."},{"key":"2615_CR38","unstructured":"Singer, U., Polyak, A., Hayes, T., Yin, X., An, J., Zhang, S., Hu, Q., Yang, H., Ashual, O., Gafni, O., and others. (2023). Make-a-video: Text-to-video generation without text-video data. In: ICLR."},{"key":"2615_CR39","unstructured":"Singer, U., Sheynin, S., Polyak, A., Ashual, O., Makarov, I., Kokkinos, F., Goyal, N., Vedaldi, A., Parikh, D., Johnson, J., and others. (2023). Text-to-4d dynamic scene generation. In: ICML."},{"key":"2615_CR40","unstructured":"Skectchfab. (2023). https:\/\/sketchfab.com\/."},{"key":"2615_CR41","unstructured":"Tang, J., Ren, J., Zhou, H., Liu, Z., & Zeng, G. (2024). Dreamgaussian: Generative gaussian splatting for efficient 3d content creation. In: ICLR."},{"key":"2615_CR42","doi-asserted-by":"crossref","unstructured":"Tang, J., Wang, T., Zhang, B., Zhang, T., Yi, R., Ma, L., & Chen, D. (2023). Make-it-3d: High-fidelity 3d creation from a single image with diffusion prior. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.02086"},{"key":"2615_CR43","doi-asserted-by":"crossref","unstructured":"Tsalicoglou, C., Manhardt, F., Tonioni, A., Niemeyer, M., & Tombari, F. (2024). Textmesh: Generation of realistic 3d meshes from text prompts. In: 3DV.","DOI":"10.1109\/3DV62453.2024.00154"},{"key":"2615_CR44","doi-asserted-by":"crossref","unstructured":"Voleti, V., Yao, C.-H., Boss, M., Letts, A., Pankratz, D., Tochilkin, D., Laforte, C., Rombach, R., & Jampani, V. (2024). Sv3d: Novel multi-view synthesis and 3d generation from a single image using latent video diffusion. In: ECCV.","DOI":"10.1007\/978-3-031-73232-4_25"},{"key":"2615_CR45","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., & Simoncelli, E.P. (2004). Image quality assessment: from error visibility to structural similarity. IEEE TIP."},{"key":"2615_CR46","unstructured":"Wang, P., Liu, L., Liu, Y., Theobalt, C., Komura, T., & Wang, W. (2021). Neus: Learning neural implicit surfaces by volume rendering for multi-view reconstruction. In: NeurIPS."},{"key":"2615_CR47","unstructured":"Wang, Z., Lu, C., Wang, Y., Bao, F., Li, C., Su, H., & Zhu, J. (2023). Prolificdreamer: High-fidelity and diverse text-to-3d generation with variational score distillation. In: NeurIPS."},{"key":"2615_CR48","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, B., Go, H., Kim, J.-Y., & Kim, C. (2024). Harmonyview: Harmonizing consistency and diversity in one-image-to-3d. In: CVPR.","DOI":"10.1109\/CVPR52733.2024.01006"},{"key":"2615_CR49","doi-asserted-by":"crossref","unstructured":"Wu, J., Gao, X., Liu, X., Shen, Z., Zhao, C., Feng, H., Liu, J., & Ding, E. (2024). Hd-fusion: Detailed text-to-3d generation leveraging multiple noise estimation. In: WACV.","DOI":"10.1109\/WACV57701.2024.00317"},{"key":"2615_CR50","doi-asserted-by":"crossref","unstructured":"Wu, G., Yi, T., Fang, J., Xie, L., Zhang, X., Wei, W., Liu, W., Tian, Q., & Xinggang, W. (2024). 4d gaussian splatting for real-time dynamic scene rendering. In: CVPR.","DOI":"10.1109\/CVPR52733.2024.01920"},{"key":"2615_CR51","doi-asserted-by":"crossref","unstructured":"Xu, D., Jiang, Y., Wang, P., Fan, Z., Wang, Y., & Wang, Z. (2023). Neurallift-360: Lifting an in-the-wild 2d photo to a 3d object with 360deg views. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.00435"},{"key":"2615_CR52","unstructured":"YU, M., Hu, W., Xing, J., & Shan, Y..(2025). Trajectorycrafter: Redirecting camera trajectory for monocular videos via diffusion models. arXiv preprint"},{"key":"2615_CR53","doi-asserted-by":"crossref","unstructured":"Yang, Z., Gao, X., Zhou, W., Jiao, S., Zhang, Y., & Jin, X. (2024). Deformable 3d gaussians for high-fidelity monocular dynamic scene reconstruction. In: CVPR.","DOI":"10.1109\/CVPR52733.2024.01922"},{"key":"2615_CR54","unstructured":"Yang, Z., Pan, Z., Gu, C., & Zhang, L. (2025). Diffusion $$^2$$: Dynamic 3d content generation via score composition of video and multi-view diffusion models. In: ICLR."},{"key":"2615_CR55","unstructured":"Yang, Z., Yang, H., Pan, Z., Zhu, X., & Zhang, L. (2024). Real-time photorealistic dynamic scene representation and rendering with 4d gaussian splatting. In: ICLR."},{"key":"2615_CR56","unstructured":"Yin, Y., Xu, D., Wang, Z., Zhao, Y., & Wei, Y. (2023). 4dgen: Grounded 4d content generation with spatial-temporal consistency. arXiv preprint."},{"key":"2615_CR57","doi-asserted-by":"crossref","unstructured":"Zeng, Y., Jiang, Y., Zhu, S., Lu, Y., Lin, Y., Zhu, H., Hu, W., Cao, X., & Yao, Y. (2024). Stag4d: Spatial-temporal anchored generative 4d gaussians. In: ECCV.","DOI":"10.1007\/978-3-031-72764-1_10"},{"key":"2615_CR58","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., & Wang, O. (2018). The unreasonable effectiveness of deep features as a perceptual metric. In: CVPR.","DOI":"10.1109\/CVPR.2018.00068"},{"key":"2615_CR59","unstructured":"Zhao, Y., Yan, Z., Xie, E., Hong, L., Li, Z., & Lee, G.H. (2023). Animate124: Animating one image to 4d dynamic scene. arXiv preprint."},{"key":"2615_CR60","unstructured":"Zhou, Y., Zhou, D., Cheng, M.-M., Feng, J., & Hou, Q. (2024). Storydiffusion: Consistent self-attention for long-range image and video generation. In: NeurIPS."},{"key":"2615_CR61","unstructured":"Zhu, J., & Zhuang, P. (204). Hifa: High-fidelity text-to-3d with advanced diffusion guidance. In: ICLR."},{"key":"2615_CR62","doi-asserted-by":"crossref","unstructured":"Zitnick, C.L., Kang, S.B., Uyttendaele, M., Winder, S., & Szeliski, R. (2004). High-quality video view interpolation using a layered representation. In: ACM TOG.","DOI":"10.1145\/1186562.1015766"},{"key":"2615_CR63","doi-asserted-by":"crossref","unstructured":"Zwicker, Pfister, Baar, V., & Gross (2001). Ewa volume splatting. In: Visualization, Vis.","DOI":"10.1145\/383259.383300"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02615-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02615-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02615-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T15:37:19Z","timestamp":1771601839000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02615-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,24]]},"references-count":63,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["2615"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02615-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,24]]},"assertion":[{"value":"22 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"14"}}