{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:02:45Z","timestamp":1777654965413,"version":"3.51.4"},"publisher-location":"Cham","reference-count":108,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729393","type":"print"},{"value":"9783031729409","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,17]],"date-time":"2024-11-17T00:00:00Z","timestamp":1731801600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,17]],"date-time":"2024-11-17T00:00:00Z","timestamp":1731801600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72940-9_18","type":"book-chapter","created":{"date-parts":[[2024,11,16]],"date-time":"2024-11-16T20:40:39Z","timestamp":1731789639000},"page":"311-330","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Cascade-Zero123: One Image to\u00a0Highly Consistent 3D with\u00a0Self-prompted Nearby Views"],"prefix":"10.1007","author":[{"given":"Yabo","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiemin","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyang","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Taoran","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaopeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingxi","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinggang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenrui","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongkai","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,17]]},"reference":[{"key":"18_CR1","unstructured":"Stable diffusion image variations. - a hugging face space by lambdalabs (2023)"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Alldieck, T., Kolotouros, N., Sminchisescu, C.: Score distillation sampling with learned manifold corrective. arXiv:2401.05293 (2024)","DOI":"10.1007\/978-3-031-73021-4_1"},{"key":"18_CR3","unstructured":"Armandpour, M., Zheng, H., Sadeghian, A., Sadeghian, A., Zhou, M.: Re-imagine the negative prompt algorithm: transform 2D diffusion into 3D, alleviate Janus problem and beyond. arXiv:2304.04968 (2023)"},{"key":"18_CR4","doi-asserted-by":"publisher","first-page":"1483","DOI":"10.1109\/TPAMI.2019.2956516","volume":"43","author":"Z Cai","year":"2019","unstructured":"Cai, Z., Vasconcelos, N.: Cascade R-CNN: high quality object detection and instance segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 43, 1483\u20131498 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the International Conference on Computer Vision (ICCV) (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Chan, E.R., et al.: GeNVS: generative novel view synthesis with 3D-aware diffusion models. In: arXiv (2023)","DOI":"10.1109\/ICCV51070.2023.00389"},{"key":"18_CR7","unstructured":"Chen, M., et al.: Sketch2NeRF: multi-view sketch-guided text-to-3D generation. arXiv:2401.14257 (2024)"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, Y., Jiao, N., Jia, K.: Fantasia3D: disentangling geometry and appearance for high-quality text-to-3D content creation. arXiv:2303.13873 (2023)","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"18_CR9","unstructured":"Chen, X., Fan, H., Girshick, R., He, K.: Improved baselines with momentum contrastive learning. arXiv (2020)"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Chen, X., Mihajlovic, M., Wang, S., Prokudin, S., Tang, S.: Morphable diffusion: 3D-consistent diffusion for single-image avatar creation. arXiv:2401.04728 (2024)","DOI":"10.1109\/CVPR52733.2024.00986"},{"key":"18_CR11","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1007\/978-3-031-20056-4_7","volume-title":"ECCV 2022","author":"Y Chen","year":"2022","unstructured":"Chen, Y., et al.: SdaE: self-distillated masked autoencoder. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13690, pp. 108\u2013124. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20056-4_7"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y., Ni, J., Jiang, N., Zhang, Y., Zhu, Y., Huang, S.: Single-view 3D scene reconstruction with high-fidelity shape and texture. arXiv:2311.00457 (2023)","DOI":"10.1109\/3DV62453.2024.00142"},{"key":"18_CR13","unstructured":"Chen, Y., et al.: 2L3: lifting imperfect generated 2D images into accurate 3D. arXiv:2401.15841 (2024)"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wang, F., Liu, H.: Text-to-3D using gaussian splatting. arXiv:2309.16585 (2023)","DOI":"10.1109\/CVPR52733.2024.02022"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Deitke, M., et al.: Objaverse-XL: a universe of 10M+ 3D objects. arXiv preprint arXiv:2307.05663 (2023)","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Deitke, M., et al.: Objaverse: a universe of annotated 3D objects. In: CVPR, pp. 13142\u201313153 (2023)","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"18_CR17","unstructured":"Dhariwal, Prafulla, , Nichol., A.: Diffusion models beat GANs on image synthesis. Adv. Neural Inf. Process. Syst. (2021)"},{"key":"18_CR18","doi-asserted-by":"publisher","unstructured":"Downs, L., et al.: Google scanned objects: a high-quality dataset of 3D scanned household items. In: 2022 International Conference on Robotics and Automation (ICRA), pp. 2553\u20132560 (2022). https:\/\/doi.org\/10.1109\/ICRA46639.2022.9811809","DOI":"10.1109\/ICRA46639.2022.9811809"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Fang, J., Wang, J., Zhang, X., Xie, L., Tian, Q.: GaussianEditor: editing 3D Gaussians delicately with text instructions. arXiv preprint arXiv:2311.16037 (2023)","DOI":"10.1109\/CVPR52733.2024.01975"},{"key":"18_CR20","first-page":"31841","volume":"35","author":"J Gao","year":"2022","unstructured":"Gao, J., et al.: GET3D: a generative model of high quality 3D textured shapes learned from images. Adv. Neural. Inf. Process. Syst. 35, 31841\u201331854 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"18_CR21","unstructured":"Gupta, A., Xiong, W., Nie, Y., Jones, I., O\u011fuz, B.: 3DGen: triplane latent diffusion for textured mesh generation. arXiv:2303.05371 (2023)"},{"key":"18_CR22","doi-asserted-by":"crossref","unstructured":"Hamdi, A., Ghanem, B., Nie\u00dfsner, M.: SPARF: large-scale learning of 3D sparse radiance fields from few input images. In: ICCV, pp. 2930\u20132940 (2023)","DOI":"10.1109\/ICCVW60793.2023.00315"},{"key":"18_CR23","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: CVPR, pp. 9729\u20139738 (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"18_CR24","unstructured":"Hu, S., et al.: HumanLiff: layer-wise 3D human generation with diffusion model. arXiv:2308.09712 (2023)"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"Huang, Z., Stojanov, S., Thai, A., Jampani, V., Rehg, J.M.: ZeroShape: regression-based zero-shot shape reconstruction. arXiv:2312.14198 (2024)","DOI":"10.1109\/CVPR52733.2024.00959"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Jain, A., Mildenhall, B., Barron, J.T., Abbeel, P., Poole, B.: Zero-shot text-guided object generation with dream fields. In: CVPR, pp. 867\u2013876 (2022)","DOI":"10.1109\/CVPR52688.2022.00094"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Jain, A., Tancik, M., Abbeel, P.: Putting NeRF on a diet: semantically consistent few-shot view synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5885\u20135894, October 2021","DOI":"10.1109\/ICCV48922.2021.00583"},{"key":"18_CR28","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. ICLR (2021)"},{"key":"18_CR29","unstructured":"Jun, H., Nichol, A.: Shap-E: generating conditional 3D implicit functions. arXiv:2305.02463 (2023)"},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Kant, Yet al.: SPAD: spatially aware multiview diffusers. arXiv:2402.05235 (2024)","DOI":"10.1109\/CVPR52733.2024.00956"},{"key":"18_CR31","doi-asserted-by":"crossref","unstructured":"Kocsis, P., Sitzmann, V., Nie\u00dfner, M.: Intrinsic image diffusion for single-view material estimation. arXiv:2312.12274 (2023)","DOI":"10.1109\/CVPR52733.2024.00497"},{"key":"18_CR32","unstructured":"Lee, D., Kim, C., Cho, M., Han, W.S.: Locality-aware generalizable implicit neural representation. In: arXiv:2310.05624 (2023)"},{"key":"18_CR33","first-page":"30923","volume":"35","author":"J Lei","year":"2022","unstructured":"Lei, J., Zhang, Y., Jia, K., et al.: TANGO: text-driven photorealistic and robust 3D stylization via lighting decomposition. Adv. Neural. Inf. Process. Syst. 35, 30923\u201330936 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"18_CR34","unstructured":"Li, H., Shi, B., Dai, W., Chen, Y., Wang, B., Sun, Y.: Hierarchical graph networks for 3D human pose estimation. arXiv:2111.11927 (2021)"},{"key":"18_CR35","unstructured":"Li, K., Wang, S., Zhang, X., Xu, Y., Xu, W., Tu, Z.: Pose recognition with cascade transformers"},{"key":"18_CR36","unstructured":"Li, S., Zanjani, F.G., Yahia, H.B., Asano, Y.M., Gall, J., Habibian, A.: Valid: variable-length input diffusion for novel view synthesis. arXiv:2312.08892 (2023)"},{"key":"18_CR37","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: Learning the 3D fauna of the web. arXiv:2401.02400 (2024)","DOI":"10.1109\/CVPR52733.2024.00931"},{"key":"18_CR38","doi-asserted-by":"crossref","unstructured":"Lin, C.H., et al.: Magic3D: high-resolution text-to-3D content creation. In: CVPR, pp. 300\u2013309 (2023)","DOI":"10.1109\/CVPR52729.2023.00037"},{"key":"18_CR39","doi-asserted-by":"crossref","unstructured":"Lin, Y., Han, H., Gong, C., Xu, Z., Zhang, Y., Li, X.: Consistent123: one image to highly consistent 3D asset using case-aware diffusion priors. arXiv:2309.17261 (2023)","DOI":"10.1145\/3664647.3680994"},{"key":"18_CR40","doi-asserted-by":"crossref","unstructured":"Liu, M., et al.: One-2-3-45++: fast single image to 3D objects with consistent multi-view generation and 3D diffusion. arXiv:2311.07885 (2023)","DOI":"10.1109\/CVPR52733.2024.00960"},{"key":"18_CR41","unstructured":"Liu, M., et\u00a0al.: One-2-3-45: any single image to 3D mesh in 45 seconds without per-shape optimization. arXiv:2306.16928 (2023)"},{"key":"18_CR42","doi-asserted-by":"crossref","unstructured":"Liu, R., Wu, R., Hoorick, B.V., Tokmakov, P., Zakharov, S., Vondrick, C.: Zero-1-to-3: Zero-shot one image to 3d object. arXiv:2303.11328 (2023)","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"18_CR43","doi-asserted-by":"publisher","unstructured":"Liu, T., Zhao, H., Yu, Y., Zhou, G., Liu, M.: Car-studio: learning car radiance fields from single-view and unlimited in-the-wild images. IEEE Robot. Autom. Lett., 2024\u20132031 (2024). https:\/\/doi.org\/10.1109\/LRA.2024.3349949","DOI":"10.1109\/LRA.2024.3349949"},{"key":"18_CR44","unstructured":"Liu, X., Kao, S.H., Chen, J., Tai, Y.W., Tang, C.K.: Deceptive-NeRF: enhancing nerf reconstruction using pseudo-observations from diffusion models. arXiv:2305.15171 (2023)"},{"key":"18_CR45","unstructured":"Liu, Y., et al.: SyncDreamer: generating multiview-consistent images from a single-view image. arXiv:2309.03453 (2023)"},{"key":"18_CR46","doi-asserted-by":"crossref","unstructured":"Long, X., et\u00a0al.: Wonder3D: single image to 3D using cross-domain diffusion. arXiv:2310.15008 (2023)","DOI":"10.1109\/CVPR52733.2024.00951"},{"key":"18_CR47","unstructured":"Luo, T., Rockwell, C., Lee, H., Johnson, J.: Scalable 3D captioning with pretrained models. arXiv:2306.07279 (2023)"},{"key":"18_CR48","unstructured":"Melas-Kyriazi, L., et al.: IM-3D: iterative multiview diffusion and reconstruction for high-quality 3D generation. arXiv:2402.08682 (2024)"},{"key":"18_CR49","doi-asserted-by":"crossref","unstructured":"Melas-Kyriazi, L., Laina, I., Rupprecht, C., Vedaldi, A.: RealFusion: 360deg reconstruction of any object from a single image. In: CVPR, pp. 8446\u20138455 (2023)","DOI":"10.1109\/CVPR52729.2023.00816"},{"key":"18_CR50","doi-asserted-by":"crossref","unstructured":"Michel, O., Bar-On, R., Liu, R., Benaim, S., Hanocka, R.: Text2mesh: text-driven neural stylization for meshes. In: CVPR, pp. 13492\u201313502 (2022)","DOI":"10.1109\/CVPR52688.2022.01313"},{"key":"18_CR51","doi-asserted-by":"crossref","unstructured":"Mohammad\u00a0Khalid, N., Xie, T., Belilovsky, E., Popa, T.: CLIP-mesh: generating textured meshes from text using pretrained image-text models. In: SIGGRAPH Asia 2022 Conference Papers, pp.\u00a01\u20138 (2022)","DOI":"10.1145\/3550469.3555392"},{"key":"18_CR52","unstructured":"Nichol, A., et al.: Glide: towards photorealistic image generation and editing with text-guided diffusion models. arXiv:2112.10741 (2021)"},{"key":"18_CR53","unstructured":"Nichol, A., Jun, H., Dhariwal, P., Mishkin, P., Chen, M.: Point-e: a system for generating 3D point clouds from complex prompts. arXiv:2212.08751 (2022)"},{"key":"18_CR54","unstructured":"Ouyang, Y., Chai, W., Ye, J., Tao, D., Zhan, Y., Wang, G.: Chasing consistency in text-to-3D generation from a single image. arXiv:2309.03599 (2023)"},{"key":"18_CR55","doi-asserted-by":"crossref","unstructured":"Paliwal, A., Nguyen, B., Tsarov, A., Kalantari, N.K.: Reshader: view-dependent highlights for single image view-synthesis. arXiv:2309.10689 (2023)","DOI":"10.1145\/3618393"},{"key":"18_CR56","unstructured":"Pan, X., Yang, Z., Bai, S., Yang, Y.: Gd$$\\hat{2}$$-NeRF: generative detail compensation via GAN and diffusion for one-shot generalizable neural radiance fields. arXiv:2401.00616 (2024)"},{"key":"18_CR57","unstructured":"Pan, Z., Yang, Z., Zhu, X., Zhang, L.: Fast dynamic 3d object generation from a single-view video. arXiv:2401.08742 (2024)"},{"key":"18_CR58","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J.: SDXL: improving latent diffusion models for high-resolution image synthesis. arXiv:2307.01952 (2023)"},{"key":"18_CR59","unstructured":"Poole, B., Jain, A., Barron, J.T., Mildenhall, B.: DreamFusion: text-to-3D using 2D diffusion. arXiv (2022)"},{"key":"18_CR60","unstructured":"Qian, G., et\u00a0al.: Magic123: one image to high-quality 3D object generation using both 2D and 3D diffusion priors. arXiv:2306.17843 (2023)"},{"key":"18_CR61","unstructured":"Qian, X., et al.: Pushing auto-regressive models for 3D shape generation at capacity and scalability. arXiv:2402.12225 (2024)"},{"key":"18_CR62","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv:2204.06125 (2022)"},{"key":"18_CR63","doi-asserted-by":"crossref","unstructured":"Roessle, B., M\u00fcller, N., Porzi, L., Bul\u00f2, S.R., Kontschieder, P., Nie\u00dfner, M.: GANeRF: leveraging discriminators to optimize neural radiance fields. In: arXiv:2306.06044. (2023)","DOI":"10.1145\/3618402"},{"key":"18_CR64","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"18_CR65","doi-asserted-by":"crossref","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E.L.: Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural Inf. Process. Syst. (2022)","DOI":"10.1145\/3528233.3530757"},{"key":"18_CR66","unstructured":"Saharia, C., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural Inf. Process. Syst. (2022)"},{"key":"18_CR67","doi-asserted-by":"crossref","unstructured":"Sanghi, A., et al.: Clip-forge: towards zero-shot text-to-shape generation. In: CVPR, pp. 18603\u201318613 (2022)","DOI":"10.1109\/CVPR52688.2022.01805"},{"key":"18_CR68","doi-asserted-by":"crossref","unstructured":"Sargent, K., et\u00a0al.: ZeroNVS: zero-shot 360-degree view synthesis from a single real image. arXiv:2310.17994 (2023)","DOI":"10.1109\/CVPR52733.2024.00900"},{"key":"18_CR69","unstructured":"Shen, Q., Yang, X., Wang, X.: Anything-3D: towards single-view anything reconstruction in the wild. arXiv:2304.10261 (2023)"},{"key":"18_CR70","unstructured":"Shi, R., et al.: Zero123++: a single image to consistent multi-view diffusion base model. arXiv:2310.15110 (2023)"},{"key":"18_CR71","unstructured":"Shi, Y., Wang, P., Ye, J., Long, M., Li, K., Yang, X.: MVDream: multi-view diffusion for 3D generation. arXiv:2308.16512 (2023)"},{"key":"18_CR72","unstructured":"Shi, Y., et al.: TOSS: high-quality text-guided novel view synthesis from a single image. arXiv:2310.10644 (2023)"},{"key":"18_CR73","unstructured":"Simon, C., He, S., Perez-Rua, J.M., Xu, M., Benhalloum, A., Xiang, T.: Hyper-VolTran: fast and generalizable one-shot image to 3D object structure via hypernetworks. arXiv:2312.16218 (2024)"},{"key":"18_CR74","unstructured":"Spiegl, B., Perin, A., Deny, S., Ilin, A.: ViewFusion: learning composable diffusion models for novel view synthesis. arXiv:2402.02906 (2024)"},{"key":"18_CR75","doi-asserted-by":"crossref","unstructured":"Tang, J., Chen, Z., Chen, X., Wang, T., Zeng, G., Liu, Z.: LGM: large multi-view gaussian model for high-resolution 3D content creation. arXiv:2402.05054 (2024)","DOI":"10.1007\/978-3-031-73235-5_1"},{"key":"18_CR76","unstructured":"Tang, J., Ren, J., Zhou, H., Liu, Z., Zeng, G.: DreamGaussian: generative Gaussian splatting for efficient 3D content creation. arXiv:2309.16653 (2023)"},{"key":"18_CR77","doi-asserted-by":"crossref","unstructured":"Tang, J., et al.: Make-it-3D: High-fidelity 3d creation from a single image with diffusion prior. arXiv:2303.14184 (2023)","DOI":"10.1109\/ICCV51070.2023.02086"},{"key":"18_CR78","doi-asserted-by":"crossref","unstructured":"Tang, S., et al.: MVDiffusion++: a dense high-resolution multi-view diffusion model for single or sparse-view 3d object reconstruction. arXiv:2402.12712 (2024)","DOI":"10.1007\/978-3-031-72640-8_10"},{"key":"18_CR79","unstructured":"Tremblay, J., et al.: RTMV: a ray-traced multi-view synthetic dataset for novel view synthesis. arXiv:2205.07058 (2022)"},{"key":"18_CR80","doi-asserted-by":"crossref","unstructured":"Vainer, S., et al.: Collaborative control for geometry-conditioned PBR image generation. arXiv:2402.05919 (2024)","DOI":"10.1007\/978-3-031-72624-8_8"},{"key":"18_CR81","doi-asserted-by":"crossref","unstructured":"Wang, C., Chai, M., He, M., Chen, D., Liao, J.: CLIP-NeRF: text-and-image driven manipulation of neural radiance fields. In: CVPR, pp. 3835\u20133844 (2022)","DOI":"10.1109\/CVPR52688.2022.00381"},{"key":"18_CR82","doi-asserted-by":"crossref","unstructured":"Wang, H., Du, X., Li, J., Yeh, R.A., Shakhnarovich, G.: Score Jacobian chaining: lifting pretrained 2D diffusion models for 3D generation. In: CVPR, pp. 12619\u201312629 (2023)","DOI":"10.1109\/CVPR52729.2023.01214"},{"key":"18_CR83","doi-asserted-by":"crossref","unstructured":"Wang, H., Du, X., Li, J., Yeh, R.A., Shakhnarovich, G.: Score Jacobian chaining: lifting pretrained 2d diffusion models for 3d generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12619\u201312629, June 2023","DOI":"10.1109\/CVPR52729.2023.01214"},{"key":"18_CR84","doi-asserted-by":"crossref","unstructured":"Wang, P., Liu, L., Liu, Y., Theobalt, C., Komura, T., Wang, W.: NeuS: learning neural implicit surfaces by volume rendering for multi-view reconstruction. arXiv:2106.10689 (2023)","DOI":"10.1109\/ICCV51070.2023.00305"},{"key":"18_CR85","unstructured":"Wang, Z., et al.: ProlificDreamer: high-fidelity and diverse text-to-3D generation with variational score distillation. arXiv:2305.16213 (2023)"},{"key":"18_CR86","unstructured":"Weng, H., et al.: Consistent123: improve consistency for one image to 3D object synthesis. arXiv:2310.08092 (2023)"},{"key":"18_CR87","unstructured":"Weng, Z., Wang, Z., Yeung, S.: ZeroAvatar: zero-shot 3D avatar generation from a single image. arXiv:2305.16411 (2023)"},{"key":"18_CR88","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, B., Go, H., Kim, J.Y., Kim, C.: HarmonyView: harmonizing consistency and diversity in one-image-to-3D. arXiv:2312.15980 (2023)","DOI":"10.1109\/CVPR52733.2024.01006"},{"key":"18_CR89","unstructured":"Wu, C.H., Chen, Y.C., Solarte, B., Yuan, L., Sun, M.: iFusion: inverting diffusion for pose-free reconstruction from sparse views. arXiv:2312.17250 (2023)"},{"key":"18_CR90","doi-asserted-by":"crossref","unstructured":"Wu, G., et al.: 4D Gaussian splatting for real-time dynamic scene rendering. arXiv preprint arXiv:2310.08528 (2023)","DOI":"10.1109\/CVPR52733.2024.01920"},{"key":"18_CR91","doi-asserted-by":"publisher","unstructured":"Wu, T., et al.: HyperDreamer: hyper-realistic 3D content generation and editing from a single image. In: SIGGRAPH Asia 2023 Conference Papers (2023). https:\/\/doi.org\/10.1145\/3610548.3618168","DOI":"10.1145\/3610548.3618168"},{"key":"18_CR92","doi-asserted-by":"crossref","unstructured":"Wu, Z., et al.: BlockFusion: expandable 3D scene generation using latent tri-plane extrapolation. arXiv:2401.17053 (2024)","DOI":"10.1145\/3658188"},{"key":"18_CR93","doi-asserted-by":"crossref","unstructured":"Xiang, J., Yang, J., Huang, B., Tong, X.: 3D-aware image generation using 2D diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2383\u20132393, October 2023","DOI":"10.1109\/ICCV51070.2023.00226"},{"key":"18_CR94","doi-asserted-by":"crossref","unstructured":"Xu, D., Jiang, Y., Wang, P., Fan, Z., Wang, Y., Wang, Z.: NeuralLift-360: lifting an in-the-wild 2D photo to a 3D object with 360deg views. In: CVPR, pp. 4479\u20134489 (2023)","DOI":"10.1109\/CVPR52729.2023.00435"},{"key":"18_CR95","unstructured":"Xu, D., et al.: AGG: amortized generative 3D Gaussians for single image to 3D. arXiv:2401.04099 (2024)"},{"key":"18_CR96","doi-asserted-by":"crossref","unstructured":"Xu, J., et al.: Dream3D: zero-shot text-to-3D synthesis using 3D shape prior and text-to-image diffusion models. In: CVPR, pp. 20908\u201320918 (2023)","DOI":"10.1109\/CVPR52729.2023.02003"},{"key":"18_CR97","unstructured":"Yang, C., L.: GaussianObject: just taking four images to get a high-quality 3D object with gaussian splatting. arXiv:2402.10259 (2024)"},{"key":"18_CR98","doi-asserted-by":"crossref","unstructured":"Yang, J., Cheng, Z., Duan, Y., Ji, P., Li, H.: ConsistNet: enforcing 3D consistency for multi-view images diffusion. arXiv:2310.10343 (2023)","DOI":"10.1109\/CVPR52733.2024.00676"},{"key":"18_CR99","doi-asserted-by":"crossref","unstructured":"Ye, J., Wang, P., Li, K., Shi, Y., Wang, H.: Consistent-1-to-3: consistent image to 3D view synthesis via geometry-aware diffusion models. arXiv:2310.03020 (2023)","DOI":"10.1109\/3DV62453.2024.00027"},{"key":"18_CR100","doi-asserted-by":"crossref","unstructured":"Ye, M., et al.: Cascade-DETR: delving into high-quality universal object detection. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00617"},{"key":"18_CR101","unstructured":"Yi, T., et al.: GaussianDreamer: fast generation from text to 3D Gaussian splatting with point cloud priors. arxiv:2310.08529 (2023)"},{"key":"18_CR102","unstructured":"Yu, K., Liu, J., Feng, M., Cui, M., Xie, X.: Boosting3D: high-fidelity image-to-3D by boosting 2D diffusion prior to 3D prior with progressive learning. arXiv:2311.13617 (2023)"},{"key":"18_CR103","doi-asserted-by":"crossref","unstructured":"Yu, Y., Zhu, S., Qin, H., Li, H.: BoostDream: efficient refining for high-quality text-to-3D generation from multi-view diffusion. arXiv:2401.16764 (2024)","DOI":"10.24963\/ijcai.2024\/598"},{"key":"18_CR104","doi-asserted-by":"crossref","unstructured":"Zeng, X., et al.: Paint3D: paint anything 3D with lighting-less texture diffusion models. arXiv:2312.13913 (2023)","DOI":"10.1109\/CVPR52733.2024.00407"},{"key":"18_CR105","doi-asserted-by":"crossref","unstructured":"Zhang, J., et al.: Repaint123: fast and high-quality one image to 3D generation with progressive controllable 2D repainting. arXiv:2312.13271 (2023)","DOI":"10.1007\/978-3-031-72698-9_18"},{"key":"18_CR106","unstructured":"Zhang, S., et al.: I2VGen-XL: high-quality image-to-video synthesis via cascaded diffusion models (2023)"},{"key":"18_CR107","unstructured":"Zhao, M., et al.: EfficientDreamer: high-fidelity and robust 3D creation via orthogonal-view diffusion prior. arXiv:2308.13223 (2023)"},{"key":"18_CR108","doi-asserted-by":"crossref","unstructured":"Zheng, X.Y., Pan, H., Guo, Y.X., Tong, X., Liu, Y.: MVD$$^2$$: efficient multiview 3D reconstruction for multiview diffusion. arXiv:2402.14253 (2024)","DOI":"10.1145\/3641519.3657403"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72940-9_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T12:17:59Z","timestamp":1733055479000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72940-9_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,17]]},"ISBN":["9783031729393","9783031729409"],"references-count":108,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72940-9_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,17]]},"assertion":[{"value":"17 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}