{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T14:03:13Z","timestamp":1760623393341,"version":"3.40.5"},"reference-count":82,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T00:00:00Z","timestamp":1734912000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T00:00:00Z","timestamp":1734912000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s11263-024-02326-x","type":"journal-article","created":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T04:35:10Z","timestamp":1734928510000},"page":"3105-3128","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["SLIDE: A Unified Mesh and Texture Generation Framework with Enhanced Geometric Control and Multi-view Consistency"],"prefix":"10.1007","volume":"133","author":[{"given":"Jinyi","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaoyang","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ben","family":"Fei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6115-5194","authenticated-orcid":false,"given":"Jiangchao","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ya","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Dai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dahua","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ying","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanfeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,23]]},"reference":[{"key":"2326_CR1","unstructured":"Anciukevicius, T., Manhardt, F., Tombari, F., & Henderson, P. (2024). Denoising diffusion via image-based rendering. Preprint retrieved from arXiv:2402.03445"},{"key":"2326_CR2","doi-asserted-by":"crossref","unstructured":"Anciukevi\u010dius, T., Xu, Z., Fisher, M., Henderson, P., Bilen, H., Mitra, N. J., & Guerrero, P. (2023). Renderdiffusion: Image diffusion for 3d reconstruction, inpainting and generation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp 12608\u201312618).","DOI":"10.1109\/CVPR52729.2023.01213"},{"key":"2326_CR3","doi-asserted-by":"crossref","unstructured":"Bokhovkin, A., Tulsiani, S., & Dai, A. (2023) Mesh2tex: Generating mesh textures from image queries. In Proceedings of the IEEE\/CVF international conference on computer vision (ICCV) (pp. 8918\u20138928)","DOI":"10.1109\/ICCV51070.2023.00819"},{"key":"2326_CR4","doi-asserted-by":"publisher","unstructured":"Cai, R., Yang, G., Averbuch-Elor, H., Hao, Z., Belongie, S., Snavely, N., & Hariharan, B. (2020). Learning gradient fields for shape generation. In A. Vedaldi, H. Bischof, T. Brox et al. (Eds.) Computer Vision-ECCV 2020-16th European conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part III, Lecture Notes in Computer Science (Vol. 12348, pp. 364\u2013381). Springer. https:\/\/doi.org\/10.1007\/978-3-030-58580-8_22","DOI":"10.1007\/978-3-030-58580-8_22"},{"key":"2326_CR5","doi-asserted-by":"crossref","unstructured":"Cao, T., Kreis, K., Fidler, S., Sharp, N., & Yin, K. (2023). Texfusion: Synthesizing 3d textures with text-guided image diffusion models. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 4169\u20134181)","DOI":"10.1109\/ICCV51070.2023.00385"},{"key":"2326_CR6","doi-asserted-by":"crossref","unstructured":"Chan, E. R., Lin, C. Z., Chan, M. A., Nagano, K., Pan, B., De Mello, S., Gallo, O., Guibas, L. J., Tremblay, J., Khamis, S., & Karras, T. (2022). Efficient geometry-aware 3d generative adversarial networks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 16123\u201316133).","DOI":"10.1109\/CVPR52688.2022.01565"},{"key":"2326_CR7","unstructured":"Chang, A. X., Funkhouser, T., Guibas, L., Hanrahan, P., Huang, Q., Li, Z., Savarese, S., Savva, M., Song, S., Su, H., & Xiao, J. (2015). Shapenet: An information-rich 3d model repository. Preprint retrieved from arXiv:1512.03012"},{"key":"2326_CR8","doi-asserted-by":"crossref","unstructured":"Chen, D. Z., Siddiqui, Y., Lee, H. Y., Tulyakov, S., & Nie\u00dfner, M. (2023a). Text2tex: Text-driven texture synthesis via diffusion models. Preprint retrieved from arXiv:2303.11396","DOI":"10.1109\/ICCV51070.2023.01701"},{"key":"2326_CR9","doi-asserted-by":"crossref","unstructured":"Chen, Q., Chen, Z., Zhou, H., & Zhang, H. (2023b) ShaDDR: Real-time example-based geometry and texture generation via 3d shape detailization and differentiable rendering. Preprint retrieved from arXiv:2306.04889","DOI":"10.1145\/3610548.3618201"},{"key":"2326_CR10","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, Y., Jiao, N., & Jia, K. (2023c) Fantasia3d: Disentangling geometry and appearance for high-quality text-to-3d content creation. Preprint retrieved from arXiv:2303.13873","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"2326_CR11","first-page":"30923","volume":"35","author":"Y Chen","year":"2022","unstructured":"Chen, Y., Chen, R., Lei, J., Zhang, Y., & Jia, K. (2022). Tango: Text-driven photorealistic and robust 3d stylization via lighting decomposition. Advances in Neural Information Processing Systems, 35, 30923\u201330936.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2326_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y., Zhang, C., Yang, X., Cai, Z., Yu, G., Yang, L., & Lin, G. (2023d). It3d: Improved text-to-3d generation with explicit view synthesis. Preprint retrieved from arXiv:2308.11473","DOI":"10.1609\/aaai.v38i2.27886"},{"key":"2326_CR13","unstructured":"Chou, G., Bahat, Y., & Heide, F. (2022). Diffusionsdf: Conditional generative modeling of signed distance functions. Preprint retrieved from arXiv:2211.13757. https:\/\/api.semanticscholar.org\/CorpusID:254017862"},{"key":"2326_CR14","doi-asserted-by":"crossref","unstructured":"Collins, J., Goel, S., Deng, K., Luthra, A., Xu, L., Gundogdu, E., Zhang, X., Vicente, T. F. Y., Dideriksen, T., Arora, H., & Guillaumin, M. (2022). Abo: Dataset and benchmarks for real-world 3d object understanding. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 21126\u201321136).","DOI":"10.1109\/CVPR52688.2022.02045"},{"key":"2326_CR15","doi-asserted-by":"crossref","unstructured":"Deitke, M., Schwenk, D., Salvador, J., Weihs, L., Michel, O., VanderBilt, E., Schmidt, L., Ehsani, K., Kembhavi, A., Farhadi, A. (2022). Objaverse: A universe of annotated 3d objects. Preprint retrieved from arXiv:2212.08051","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"2326_CR16","doi-asserted-by":"crossref","unstructured":"Deitke, M., Schwenk, D., Salvador, J., Weihs, L., Michel, O., VanderBilt, E., Schmidt, L., Ehsani, K., Kembhavi, A., & Farhadi, A. (2023). Objaverse: A universe of annotated 3d objects. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 13142\u201313153).","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"2326_CR17","first-page":"31841","volume":"35","author":"J Gao","year":"2022","unstructured":"Gao, J., Shen, T., Wang, Z., Chen, W., Yin, K., Li, D., Litany, O., Gojcic, Z., & Fidler, S. (2022). Get3d: A generative model of high quality 3d textured shapes learned from images. Advances in Neural Information Processing Systems, 35, 31841\u201331854.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2326_CR18","unstructured":"Gupta, A., Xiong, W., Nie, Y., Jones, I., & O\u011fuz, B. (2023). 3dgen: Triplane latent diffusion for textured mesh generation. Preprint retrieved from arXiv:2303.05371"},{"key":"2326_CR19","doi-asserted-by":"crossref","unstructured":"Henderson, P., Tsiminaki, V., & Lampert, C. H. (2020). Leveraging 2d data to learn textured 3d mesh generation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 7498\u20137507).","DOI":"10.1109\/CVPR42600.2020.00752"},{"key":"2326_CR20","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020a). Denoising diffusion probabilistic models. Preprint retrieved from arXiv:2006.11239"},{"key":"2326_CR21","unstructured":"Ho, J., Jain, A., & Abbeel, P (2020b) Denoising diffusion probabilistic models. In: H. Larochelle, M. Ranzato, R. Hadsell, et al. (Eds.) Advances in neural information processing systems 33: Annual conference on neural information processing systems 2020, NeurIPS 2020, December 6\u201312, 2020, virtual. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/4c5bcfec8584af0d967f1ab10179ca4b-Abstract.html"},{"key":"2326_CR22","doi-asserted-by":"crossref","unstructured":"H\u00f6llein, L., Bo\u017ei\u010d, A., M\u00fcller, N., Novotny, D., Tseng, H.Y., Richardt, C., Zollh\u00f6fer, M., & Nie\u00dfner, M. (2024). Viewdiff: 3d-consistent image generation with text-to-image models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 5043\u20135052).","DOI":"10.1109\/CVPR52733.2024.00482"},{"key":"2326_CR23","doi-asserted-by":"crossref","unstructured":"Hong, F., Zhang, M., Pan, L., Cai, Z., Yang, L., & Liu, Z. (2022). Avatarclip: Zero-shot text-driven generation and animation of 3d avatars. Preprint retrieved from arXiv:2205.08535","DOI":"10.1145\/3528223.3530094"},{"key":"2326_CR24","doi-asserted-by":"crossref","unstructured":"Huang, J., Thies, J., Dai, A., Kundu, A., Jiang, C., Guibas, L.J., Nie\u00dfner, M., & Funkhouser, T. (2020). Adversarial texture optimization from rgb-d scans. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 1559\u20131568).","DOI":"10.1109\/CVPR42600.2020.00163"},{"key":"2326_CR25","unstructured":"Jun, H., & Nichol, A. (2023). Shap-e: Generating conditional 3d implicit functions. Preprint retrieved from arXiv:2305.02463"},{"key":"2326_CR26","doi-asserted-by":"crossref","unstructured":"Karnewar, A., Mitra, N. J., Vedaldi, A., & Novotny, D. (2023). Holofusion: Towards photo-realistic 3d generative modeling. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 22976\u201322985).","DOI":"10.1109\/ICCV51070.2023.02100"},{"key":"2326_CR27","doi-asserted-by":"crossref","unstructured":"Kopf, J., Fu, C. W., Cohen-Or, D., Deussen, O., Lischinski, D., & Wong, T. T. (2007). Solid texture synthesis from 2d exemplars. In ACM SIGGRAPH 2007 papers (p 2\u2013es).","DOI":"10.1145\/1275808.1276380"},{"issue":"3","key":"2326_CR28","doi-asserted-by":"publisher","first-page":"541","DOI":"10.1145\/1141911.1141921","volume":"25","author":"S Lefebvre","year":"2006","unstructured":"Lefebvre, S., & Hoppe, H. (2006). Appearance-space texture synthesis. ACM Transactions on Graphics (TOG), 25(3), 541\u2013548.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2326_CR29","doi-asserted-by":"crossref","unstructured":"Li, M., Duan, Y., Zhou, J., & Lu, J. (2022a). Diffusion-sdf: Text-to-shape via voxelized diffusion. In 2023 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 12642\u201312651). https:\/\/api.semanticscholar.org\/CorpusID:254366593","DOI":"10.1109\/CVPR52729.2023.01216"},{"key":"2326_CR30","doi-asserted-by":"crossref","unstructured":"Li, Y., Dou, Y., Shi, Y., Lei, Y., Chen, X., Zhang, Y., Zhou, P., & Ni, B. (2023). Focaldreamer: Text-driven 3d editing via focal-fusion assembly. Preprint retrieved from arXiv:2308.10608","DOI":"10.1609\/aaai.v38i4.28113"},{"key":"2326_CR31","doi-asserted-by":"crossref","unstructured":"Li, Y., Upadhyay, U., Slim, H., Abdelreheem, A., Prajapati, A., Pothigara, S., Wonka, P., & Elhoseiny, M. (2022b). 3d compat: Composition of materials on parts of 3d things. In European conference on computer vision (pp. 110\u2013127). Springer.","DOI":"10.1007\/978-3-031-20074-8_7"},{"key":"2326_CR32","doi-asserted-by":"crossref","unstructured":"Lin, C. H., Gao, J., Tang, L., Takikawa, T., Zeng, X., Huang, X., Kreis, K., Fidler, S., Liu, M. Y., Lin, T. Y. (2023). Magic3d: High-resolution text-to-3d content creation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 300\u2013309).","DOI":"10.1109\/CVPR52729.2023.00037"},{"key":"2326_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Y., Xie, M., Liu, H., & Wong, T. T.(2023a). Text-guided texturing by synchronized multi-view diffusion. Preprint retrieved from arXiv:2311.12891","DOI":"10.1145\/3680528.3687621"},{"key":"2326_CR34","unstructured":"Liu, Z., Feng, Y., Black, M. J., Nowrouzezahrai, D., Paull, L., & Liu, W. (2023b). Meshdiffusion: Score-based generative 3d mesh modeling. Preprint retrieved from arXiv:2303.08133, https:\/\/api.semanticscholar.org\/CorpusID:257505014"},{"key":"2326_CR35","doi-asserted-by":"crossref","unstructured":"Lorensen, W. E., & Cline, H. E. (1987). Marching cubes: A high resolution 3d surface construction algorithm. In Proceedings of the 14th annual conference on Computer graphics and interactive techniques. https:\/\/api.semanticscholar.org\/CorpusID:15545924","DOI":"10.1145\/37401.37422"},{"key":"2326_CR36","doi-asserted-by":"crossref","unstructured":"Luo, S., & Hu, W. (2021a). Diffusion probabilistic models for 3d point cloud generation. In 2021 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 2836\u20132844). https:\/\/api.semanticscholar.org\/CorpusID:232092778","DOI":"10.1109\/CVPR46437.2021.00286"},{"key":"2326_CR37","doi-asserted-by":"crossref","unstructured":"Luo, S., & Hu, W. (2021b). Diffusion probabilistic models for 3d point cloud generation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 2837\u20132845).","DOI":"10.1109\/CVPR46437.2021.00286"},{"key":"2326_CR38","unstructured":"Luo, T., Rockwell, C., Lee, H., & Johnson, J. (2023). Scalable 3d captioning with pretrained models. Preprint retrieved from arXiv:2306.07279"},{"key":"2326_CR39","unstructured":"Lyu, Z., Kong, Z., Xu, X., Pan, L., & Lin, D. (2021). A conditional point diffusion-refinement paradigm for 3d point cloud completion. Preprint retrieved from arXiv:2112.03530"},{"key":"2326_CR40","doi-asserted-by":"crossref","unstructured":"Lyu, Z., Wang, J., An, Y., Zhang, Y., Lin, D., & Dai, B. (2023). Controllable mesh generation through sparse latent point diffusion models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 271\u2013280).","DOI":"10.1109\/CVPR52729.2023.00034"},{"key":"2326_CR41","doi-asserted-by":"crossref","unstructured":"Ma, Y., Zhang, X., Sun, X., Ji, J., Wang, H., Jiang, G., Zhuang, W., & Ji, R. (2023). X-mesh: Towards fast and accurate text-driven 3d stylization via dynamic textual guidance. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 2749\u20132760).","DOI":"10.1109\/ICCV51070.2023.00258"},{"key":"2326_CR42","doi-asserted-by":"crossref","unstructured":"Mescheder, L., Oechsle, M., Niemeyer, M., Nowozin, S., & Geiger, A. (2018). Occupancy networks: Learning 3d reconstruction in function space. In 2019 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 4455\u20134465). https:\/\/api.semanticscholar.org\/CorpusID:54465161","DOI":"10.1109\/CVPR.2019.00459"},{"key":"2326_CR43","doi-asserted-by":"crossref","unstructured":"Metzer, G., Richardson, E., Patashnik, O., Giryes, R., & Cohen-Or, D. (2023). Latent-nerf for shape-guided generation of 3d shapes and textures. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 12663\u201312673).","DOI":"10.1109\/CVPR52729.2023.01218"},{"key":"2326_CR44","doi-asserted-by":"crossref","unstructured":"Michel, O., Bar-On, R., Liu, R., Benaim, S., & Hanocka, R. (2022). Text2mesh: Text-driven neural stylization for meshes. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 13492\u201313502).","DOI":"10.1109\/CVPR52688.2022.01313"},{"key":"2326_CR45","doi-asserted-by":"crossref","unstructured":"Mohammad\u00a0Khalid, N., Xie, T., Belilovsky, E., & Popa, T. (2022). Clip-mesh: Generating textured meshes from text using pretrained image-text models. In SIGGRAPH Asia 2022 conference papers (pp. 1\u20138).","DOI":"10.1145\/3550469.3555392"},{"key":"2326_CR46","unstructured":"Nam, G., Khlifi, M., Rodriguez, A., Tono, A., Zhou, L., & Guerrero, P. (2022). 3d-ldm: Neural implicit 3d shape generation with latent diffusion models. Preprint retrieved from arXiv:2212.00842, https:\/\/api.semanticscholar.org\/CorpusID:254220714"},{"key":"2326_CR47","unstructured":"Nichol, A., Jun, H., Dhariwal, P., Mishkin, P., & Chen, M. (2022a). Point-e: A system for generating 3d point clouds from complex prompts. Preprint retrieved from arXiv:2212.08751, https:\/\/api.semanticscholar.org\/CorpusID:254854214"},{"key":"2326_CR48","unstructured":"Nichol, A., Jun, H., Dhariwal, P., Mishkin, P., & Chen, M. (2022b). Point-e: A system for generating 3d point clouds from complex prompts. Preprint retrieved from arXiv:2212.08751"},{"key":"2326_CR49","doi-asserted-by":"crossref","unstructured":"Oechsle, M., Mescheder, L., Niemeyer, M., Strauss, T., & Geiger, A. (2019). Texture fields: Learning texture representations in function space. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 4531\u20134540).","DOI":"10.1109\/ICCV.2019.00463"},{"key":"2326_CR50","unstructured":"Pan, Z., Lu, J., Zhu, X., & Zhang, L. (2023). Enhancing high-resolution 3d generation through pixel-wise gradient clipping. Preprint retrieved from arXiv:2310.12474"},{"key":"2326_CR51","doi-asserted-by":"crossref","unstructured":"Park, J. J., Florence, P., Straub, J., Newcombe, R., & Lovegrove, S. (2019). Deepsdf: Learning continuous signed distance functions for shape representation. In 2019 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 165\u2013174). https:\/\/api.semanticscholar.org\/CorpusID:58007025","DOI":"10.1109\/CVPR.2019.00025"},{"key":"2326_CR52","first-page":"13032","volume":"34","author":"S Peng","year":"2021","unstructured":"Peng, S., Jiang, C., Liao, Y., Niemeyer, M., Pollefeys, M., & Geiger, A. (2021). Shape as points: A differentiable Poisson solver. Advances in Neural Information Processing Systems, 34, 13032\u201313044.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2326_CR53","unstructured":"Peng, S., Jiang, C., Liao, Y., Niemeyer, M., Pollefeys, M., & Geiger, A. (2021b). Shape as points: A differentiable Poisson solver. In Neural information processing systems. https:\/\/api.semanticscholar.org\/CorpusID:235358422"},{"key":"2326_CR54","unstructured":"Poole, B., Jain, A., Barron, JT., & Mildenhall, B. (2022). Dreamfusion: Text-to-3d using 2d diffusion. Preprint retrieved from arXiv:2209.14988"},{"key":"2326_CR55","unstructured":"Qi, C. R., Yi, L., Su, H., & Guibas, L. J. (2017). Pointnet++: Deep hierarchical feature learning on point sets in a metric space. In: I. Guyon, U. V. Luxburg, S. Bengio, et al. (Eds.) Advances in Neural Information Processing Systems (Vol. 30). Curran Associates, Inc., https:\/\/proceedings.neurips.cc\/paper\/2017\/file\/d8bf84be3800d12f74d8b05e9b89836f-Paper.pdf"},{"key":"2326_CR56","unstructured":"Qian, G., Mai, J., Hamdi, A., Ren, J., Siarohin, A., Li, B., Lee, H.Y., Skorokhodov, I., Wonka, P., Tulyakov, S., & Ghanem, B. (2023). Magic123: One image to high-quality 3d object generation using both 2d and 3d diffusion priors. Preprint retrieved from arXiv:2306.17843"},{"key":"2326_CR57","doi-asserted-by":"crossref","unstructured":"Raj, A., Kaza, S., Poole, B., Niemeyer, M., Ruiz, N., Mildenhall, B., Zada, S., Aberman, K., Rubinstein, M., Barron, J., & Li, Y. (2023). Dreambooth3d: Subject-driven text-to-3d generation. Preprint retrieved from arXiv:2303.13508","DOI":"10.1109\/ICCV51070.2023.00223"},{"key":"2326_CR58","doi-asserted-by":"crossref","unstructured":"Rakotosaona, M.J., Manhardt, F., Arroyo, D.M., Niemeyer, M., Kundu, A., & Tombari, F. (2024). Nerfmeshing: Distilling neural radiance fields into geometrically-accurate 3d meshes. In 2024 international conference on 3D vision (3DV) (pp. 1156\u20131165). IEEE.","DOI":"10.1109\/3DV62453.2024.00093"},{"key":"2326_CR59","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. Preprint retrieved from arXiv:2204.061251(2), 3."},{"key":"2326_CR60","doi-asserted-by":"crossref","unstructured":"Richardson, E., Metzer, G., Alaluf, Y., Giryes, R., & Cohen-Or, D. (2023). Texture: Text-guided texturing of 3d shapes. Preprint retrieved from arXiv:2302.01721","DOI":"10.1145\/3588432.3591503"},{"key":"2326_CR61","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 10684\u201310695).","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2326_CR62","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., & Ho, J. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems, 35, 36479\u201336494."},{"key":"2326_CR63","first-page":"6087","volume":"34","author":"T Shen","year":"2021","unstructured":"Shen, T., Gao, J., Yin, K., Liu, M. Y., & Fidler, S. (2021). Deep marching tetrahedra: A hybrid representation for high-resolution 3d shape synthesis. Advances in Neural Information Processing Systems, 34, 6087\u20136101.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2326_CR64","unstructured":"Shi, Y., Wang, P., Ye, J., Long, M., Li, K., & Yang, X. (2023). Mvdream: Multi-view diffusion for 3d generation. Preprint retrieved from arXiv:2308.16512"},{"key":"2326_CR65","doi-asserted-by":"crossref","unstructured":"Shue, J.R., Chan, E.R., Po, R., Ankner, Z., Wu, J., & Wetzstein, G. (2022). 3d neural field generation using triplane diffusion. In 2023 IEEE\/CVF conference on computer vision and pattern recognition (CVPR) (pp. 20875\u201320886). https:\/\/api.semanticscholar.org\/CorpusID:254095843","DOI":"10.1109\/CVPR52729.2023.02000"},{"key":"2326_CR66","doi-asserted-by":"crossref","unstructured":"Siddiqui, Y., Thies, J., Ma, F., Shan, Q., Nie\u00dfner, M., & Dai, A. (2022). Texturify: Generating textures on 3d shape surfaces. In European Conference on Computer Vision (pp. 72\u201388). Springer.","DOI":"10.1007\/978-3-031-20062-5_5"},{"key":"2326_CR67","unstructured":"Song, J., Meng, C., & Ermon, S. (2020) Denoising diffusion implicit models. In International Conference on Learning Representations."},{"key":"2326_CR68","doi-asserted-by":"crossref","unstructured":"Szymanowicz, S., Rupprecht, C., & Vedaldi, A. (2023) Viewset diffusion:(0-) image-conditioned 3d generative models from 2d data. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 8863\u20138873).","DOI":"10.1109\/ICCV51070.2023.00814"},{"key":"2326_CR69","doi-asserted-by":"crossref","unstructured":"Tang, J., Wang, T., Zhang, B., Zhang, T., Yi, R., Ma, L., & Chen, D. (2023). Make-it-3d: High-fidelity 3d creation from a single image with diffusion prior. Preprint retrieved from arXiv:2303.14184","DOI":"10.1109\/ICCV51070.2023.02086"},{"key":"2326_CR70","doi-asserted-by":"crossref","unstructured":"Turk, G. (2001). Texture synthesis on surfaces. In Proceedings of the 28th annual conference on Computer graphics and interactive techniques (pp. 347\u2013354).","DOI":"10.1145\/383259.383297"},{"key":"2326_CR71","unstructured":"Wei, L. Y., Lefebvre, S., Kwatra, V., & Turk, G. (2009). State of the art in example-based texture synthesis. Eurographics 2009, State of the Art Report, EG-STAR (pp. 93\u2013117)."},{"key":"2326_CR72","doi-asserted-by":"crossref","unstructured":"Wei, L. Y., & Levoy, M. (2001). Texture synthesis over arbitrary manifold surfaces. In Proceedings of the 28th annual conference on Computer graphics and interactive techniques (pp. 355\u2013360).","DOI":"10.1145\/383259.383298"},{"key":"2326_CR73","doi-asserted-by":"crossref","unstructured":"Yang, B., Dong, W., Ma, L., Hu, W., Liu, X., Cui, Z., & Ma, Y. (2023). Dreamspace: Dreaming your room space with text-driven panoramic texture propagation. Preprint retrieved from arXiv:2310.13119","DOI":"10.1109\/VR58804.2024.00085"},{"key":"2326_CR74","doi-asserted-by":"crossref","unstructured":"Yang, G., Huang, X., Hao, Z., Liu, M.Y., Belongie, S., & Hariharan, B. (2019). Pointflow: 3d point cloud generation with continuous normalizing flows. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 4541\u20134550).","DOI":"10.1109\/ICCV.2019.00464"},{"key":"2326_CR75","unstructured":"Zeng, X., Vahdat, A., Williams, F., Gojcic, Z., Litany, O., Fidler, S., & Kreis, K. (2022a). Lion: Latent point diffusion models for 3d shape generation. Preprint retrieved from arXiv:2210.06978. https:\/\/api.semanticscholar.org\/CorpusID:252872881"},{"key":"2326_CR76","unstructured":"Zeng, X., Vahdat, A., Williams, F., Gojcic, Z., Litany, O., Fidler, S., & Kreis, K. (2022b). Lion: Latent point diffusion models for 3d shape generation. Preprint retrieved from arXiv:2210.06978"},{"key":"2326_CR77","doi-asserted-by":"crossref","unstructured":"Zheng, X., Liu, Y., Wang, P., & Tong, X. (2022). Sdf-stylegan: Implicit sdf-based stylegan for 3d shape generation. In Computer Graphics Forum (pp. 52\u201363). Wiley Online Library.","DOI":"10.1111\/cgf.14602"},{"key":"2326_CR78","doi-asserted-by":"crossref","unstructured":"Zheng, X. Y., Pan, H., Wang, P. S., Tong, X., Liu, Y., & Shum, H. Y. (2023). Locally attentional sdf diffusion for controllable 3d shape generation. ACM Transactions on Graphics (TOG), 42, 1\u201313.","DOI":"10.1145\/3592103"},{"key":"2326_CR79","doi-asserted-by":"crossref","unstructured":"Zhou, L., Du, Y., & Wu, J. (2021a) 3d shape generation and completion through point-voxel diffusion. In 2021 IEEE\/CVF international conference on computer vision (ICCV) (pp. 5806\u20135815). https:\/\/api.semanticscholar.org\/CorpusID:233182041","DOI":"10.1109\/ICCV48922.2021.00577"},{"key":"2326_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, L., Du, Y., & Wu, J. (2021b) 3d shape generation and completion through point-voxel diffusion. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 5826\u20135835).","DOI":"10.1109\/ICCV48922.2021.00577"},{"key":"2326_CR81","doi-asserted-by":"crossref","unstructured":"Zhou, Q. Y., & Koltun, V. (2014). Color map optimization for 3d reconstruction with consumer depth cameras. ACM Transactions on Graphics (ToG), 33(4), 1\u201310.","DOI":"10.1145\/2601097.2601134"},{"key":"2326_CR82","doi-asserted-by":"crossref","unstructured":"Zhuang, J., Wang, C., Lin, L., Liu, L., & Li, G. (2023). Dreameditor: Text-driven 3d scene editing with neural fields. Preprint retrieved from arXiv:2306.13455","DOI":"10.1145\/3610548.3618190"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02326-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02326-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02326-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,10]],"date-time":"2025-05-10T06:54:18Z","timestamp":1746860058000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02326-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,23]]},"references-count":82,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["2326"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02326-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"type":"print","value":"0920-5691"},{"type":"electronic","value":"1573-1405"}],"subject":[],"published":{"date-parts":[[2024,12,23]]},"assertion":[{"value":"1 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}