{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T14:19:51Z","timestamp":1775744391401,"version":"3.50.1"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s00371-026-04453-7","type":"journal-article","created":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T18:34:05Z","timestamp":1774550045000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhanced text-driven 3D stylization: local consistency and hierarchical encoding"],"prefix":"10.1007","volume":"42","author":[{"given":"Jiayi","family":"Bu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huihuang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yichun","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiajia","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,26]]},"reference":[{"key":"4453_CR1","unstructured":"Poole, B., Jain, A., Barron, J.T., Mildenhall, B.: Dreamfusion: text-to-3d using 2d diffusion. Preprint at arXiv:2209.14988. (2022)"},{"key":"4453_CR2","doi-asserted-by":"crossref","unstructured":"Lin, C.-H., Gao, J., Tang, L., Takikawa, T., Zeng, X., Huang, X., Geng, Z., Johnson, J., Shih, K., Lin, T.-Y.: Magic3d: high-resolution text-to-3d content creation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 300\u2013309 (2023)","DOI":"10.1109\/CVPR52729.2023.00037"},{"key":"4453_CR3","doi-asserted-by":"publisher","first-page":"30923","DOI":"10.52202\/068431-2242","volume":"35","author":"Y Chen","year":"2022","unstructured":"Chen, Y., Chen, R., Lei, J., Zhang, Y., Jia, K.: Tango: Text-driven photorealistic and robust 3d stylization via lighting decomposition. Adv. Neural. Inf. Process. Syst. 35, 30923\u201330936 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4453_CR4","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J. et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"4453_CR5","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural Inf. Process. Syst. (NeurIPS) 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)"},{"key":"4453_CR6","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"4453_CR7","doi-asserted-by":"crossref","unstructured":"Michel, O., Bar-On, R., Liu, R., Benaim, S., Hanocka, R.: Text2mesh: text-driven neural stylization for meshes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13492\u201313502 (2022)","DOI":"10.1109\/CVPR52688.2022.01313"},{"key":"4453_CR8","doi-asserted-by":"crossref","unstructured":"Sanghi, A.: Clip-mesh: generating textured meshes from text using pretrained image-text models. In: ACM SIGGRAPH Asia Conference Papers, pp. 1\u20138 (2022)","DOI":"10.1145\/3550469.3555392"},{"key":"4453_CR9","doi-asserted-by":"crossref","unstructured":"Ma, Y., Zhang, X., Sun, X., Ji, J., Wang, H., Jiang, G., Zhuang, W., Ji, R.: X-mesh: towards fast and accurate text-driven 3d stylization via dynamic textual guidance. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2749\u20132760 (2023)","DOI":"10.1109\/ICCV51070.2023.00258"},{"issue":"4","key":"4453_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3528223.3530164","volume":"41","author":"R Gal","year":"2022","unstructured":"Gal, R., Patashnik, O., Maron, H., Bermano, A.H., Chechik, G., Cohen-Or, D.: Stylegan-nada: clip-guided domain adaptation of image generators. ACM Trans. Graph. (TOG) 41(4), 1\u201313 (2022)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"4453_CR11","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-Or, D.: Prompt-to-prompt image editing with cross attention control. Preprint at arXiv:2208.01626 (2022)"},{"key":"4453_CR12","doi-asserted-by":"crossref","unstructured":"Richardson, E., Metzer, G., Alaluf, Y., Giryes, R., Cohen-Or, D.: Texture: text-guided texturing of 3d shapes. In: ACM SIGGRAPH 2023 Conference Proceedings, pp. 1\u201311 (2023)","DOI":"10.1145\/3588432.3591503"},{"key":"4453_CR13","doi-asserted-by":"crossref","unstructured":"Metzer, G., Richardson, E., Patashnik, O., Giryes, R., Cohen-Or, D.: Latent-nerf for shape-guided generation of 3d shapes and textures. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12663\u201312673 (2023)","DOI":"10.1109\/CVPR52729.2023.01218"},{"issue":"5","key":"4453_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3326362","volume":"38","author":"Y Wang","year":"2019","unstructured":"Wang, Y., Sun, Y., Liu, Z., Sarma, S.E., Bronstein, M.M., Solomon, J.M.: Dynamic graph cnn for learning on point clouds. ACM Trans. Graph. (Tog) 38(5), 1\u201312 (2019)","journal-title":"ACM Trans. Graph. (Tog)"},{"key":"4453_CR15","unstructured":"Thomas, N., Smidt, T., Kearnes, S., Yang, L., Li, L., Kohlhoff, K., Riley, P.: Tensor field networks: rotation-and translation-equivariant neural networks for 3d point clouds. Preprint at arXiv:1802.08219, (2018)"},{"issue":"1","key":"4453_CR16","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: representing scenes as neural radiance fields for view synthesis. Commun. ACM 65(1), 99\u2013106 (2021)","journal-title":"Commun. ACM"},{"key":"4453_CR17","doi-asserted-by":"crossref","unstructured":"Chibane, J., Alldieck, T., Pons-Moll, G.: Implicit functions in feature space for 3d shape reconstruction and completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6970\u20136981 (2020)","DOI":"10.1109\/CVPR42600.2020.00700"},{"issue":"5","key":"4453_CR18","doi-asserted-by":"publisher","first-page":"13","DOI":"10.1111\/cgf.12693","volume":"34","author":"D Boscaini","year":"2015","unstructured":"Boscaini, D., Masci, J., Melzi, S., Bronstein, M.M., Castellani, U., Vandergheynst, P.: Learning class-specific descriptors for deformable shapes using localized spectral convolutional networks. Comput. Graph. Forum 34(5), 13\u201323 (2015)","journal-title":"Comput. Graph. Forum"},{"issue":"4","key":"4453_CR19","first-page":"1","volume":"42","author":"J Huang","year":"2023","unstructured":"Huang, J., Zhang, H., Yi, L., Funkhouser, T., Nie\u00dfner, M., Guibas, L.: Neural wavelet-domain diffusion for 3d shape generation and inversion. ACM Trans. Graph. (TOG) 42(4), 1\u201313 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"4453_CR20","doi-asserted-by":"crossref","unstructured":"Kirstain, Y., Polyak, A., Singer, U., Matiana, S., Penna, J., Levy, O.: Pick-a-pic: an open dataset of user preferences for text-to-image generation. Preprint at arXiv:2305.01569, (2023)","DOI":"10.52202\/075280-1594"},{"key":"4453_CR21","unstructured":"Kania, K., Yi, K., Kolkin, N., Tagliasacchi, A.: Doodle it yourself: class agnostic 3d reconstruction via scribbles. Preprint at arXiv:2211.12342, (2022)"},{"key":"4453_CR22","unstructured":"Liao, T., Zhang, X., Zhang, Y., Wang, M., Liu, Y.: Layoutdiffuse: adapting foundational diffusion models for layout-to-image generation. Preprint at arXiv:2302.08908, (2023)"},{"key":"4453_CR23","unstructured":"Jun, H., Nichol, A.: Shap-e: Generating conditional 3d implicit functions. Preprint at arXiv:2305.02463, (2023)"},{"key":"4453_CR24","unstructured":"Liu, Z., Feng, Y., Black, M.J., Nowrouzezahrai, D., Paull, L., Liu, W.: Meshdiffusion: score-based generative 3d mesh modeling. Preprint at arXiv:2303.08133, (2023)"},{"key":"4453_CR25","doi-asserted-by":"crossref","unstructured":"Wang, C., Chai, M., He, M., Chen, D., Liao, J.: Clip-nerf: text-and-image driven manipulation of neural radiance fields. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3835\u20133844 (2022)","DOI":"10.1109\/CVPR52688.2022.00381"},{"key":"4453_CR26","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00371-025-04047-9","volume":"41","author":"W Gao","year":"2025","unstructured":"Gao, W., Li, X., Liu, C., Wang, J., Yu, D.: Disentangled text-driven stylization of 3d faces via directional clip losses. Vis. Comput. 41, 1\u201316 (2025)","journal-title":"Vis. Comput."},{"key":"4453_CR27","unstructured":"Chen, Y., Chen, R., Lei, J., Zhang, Y., Jia, K.: 3dstyle-diffusion: pursuing fine-grained text-driven 3d stylization with 2d diffusion models. In: Proceedings of the 31st ACM International Conference on Multimedia (2023)"},{"key":"4453_CR28","unstructured":"Zhao, R., Wang, Z., Wang, Y., Zhou, Z., Zhu, J.: Flexidreamer: single image-to-3d generation with flexicubes. Preprint at arXiv:2404.00987 (2024)"},{"key":"4453_CR29","doi-asserted-by":"crossref","unstructured":"Ma, Z., Liang, X., Wu, R., Zhu, X., Lei, Z., Zhang, L.: Progressive rendering distillation: adapting stable diffusion for instant text-to-mesh generation without 3d data. (2025)","DOI":"10.1109\/CVPR52734.2025.01031"},{"key":"4453_CR30","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, Y., Jiao, N., Jia, K.: Fantasia3d: disentangling geometry and appearance for high-quality text-to-3d content creation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 22246\u201322256 (2023)","DOI":"10.1109\/ICCV51070.2023.02033"},{"issue":"4","key":"4453_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592433","volume":"42","author":"B Kerbl","year":"2023","unstructured":"Kerbl, B., Kopanas, G., Leimk\u00fchler, T., Drettakis, G.: 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graph. 42(4), 1\u201314 (2023)","journal-title":"ACM Trans. Graph."},{"key":"4453_CR32","unstructured":"Tang, J., Ren, J., Zhou, H., Liu, Z., Zeng, G.: Dreamgaussian: generative gaussian splatting for efficient 3d content creation. Preprint at arXiv:2309.16653 (2023)"},{"key":"4453_CR33","doi-asserted-by":"crossref","unstructured":"Liang, Y., Yang, X., Lin, J., Li, H., Xu, X., Chen, Y.: Luciddreamer: towards high-fidelity text-to-3d generation via interval score matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6517\u20136526 (2024)","DOI":"10.1109\/CVPR52733.2024.00623"},{"key":"4453_CR34","unstructured":"Ma, Y., Fan, Y., Ji, J., Wang, H., Sun, X., Jiang, G., Shu, An., Ji, R.: X-dreamer: creating high-quality 3d content by bridging the domain gap between text-to-2d and text-to-3d generation. Preprint at arXiv:2312.00085, (2023)"},{"key":"4453_CR35","first-page":"7537","volume":"33","author":"M Tancik","year":"2020","unstructured":"Tancik, M., Srinivasan, P., Mildenhall, B., Fridovich-Keil, S., Raghavan, N., Singhal, U., Ramamoorthi, R., Barron, J., Ng, R.: Fourier features let networks learn high frequency functions in low dimensional domains. Adv. Neural. Inf. Process. Syst. 33, 7537\u20137547 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4453_CR36","unstructured":"Rahaman, N., Baratin, A., Arpit, D., Draxler, F., Lin, M., Hamprecht, F., Bengio, Y., Courville, A.: On the spectral bias of neural networks. In: International Conference on Machine Learning, pp. 5301\u20135310. PMLR, (2019)"},{"key":"4453_CR37","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gao, Z., Wang, Y., Xiong, R., Zhang, Y.: Adaptive wavelet-positional encoding for high-frequency information learning in implicit neural representation. In: Proceedings of the AAAI Conference on Artificial Intelligence 39, 10430\u201310438 (2025)","DOI":"10.1609\/aaai.v39i10.33132"},{"key":"4453_CR38","doi-asserted-by":"crossref","unstructured":"Barron, J.T., Mildenhall, B., Tancik, M., Hedman, P., Martin-Brualla, R., Srinivasan, P.P.: Mip-nerf: a multiscale representation for anti-aliasing neural radiance fields. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5855\u20135864 (2021)","DOI":"10.1109\/ICCV48922.2021.00580"},{"key":"4453_CR39","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3d classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"4453_CR40","unstructured":"Qi, Charles\u00a0Ruizhongtai, Yi, Li, Su, Hao, Guibas, Leonidas\u00a0J.: Pointnet++: Deep hierarchical feature learning on point sets in a metric space. Advances in neural information processing systems, 30, (2017)"},{"issue":"4","key":"4453_CR41","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3306346.3322959","volume":"38","author":"R Hanocka","year":"2019","unstructured":"Hanocka, R., Hertz, A., Fish, N., Giryes, R., Fleishman, S., Cohen-Or, D.: Meshcnn: a network with an edge. ACM Trans. Graph. (ToG) 38(4), 1\u201312 (2019)","journal-title":"ACM Trans. Graph. (ToG)"},{"key":"4453_CR42","doi-asserted-by":"crossref","unstructured":"Gong, S., Chen, L., Bronstein, M., Zafeiriou, S.: Spiralnet++: a fast and highly efficient mesh convolution operator. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00509"},{"key":"4453_CR43","doi-asserted-by":"crossref","unstructured":"Feng, Y., Feng, Y., You, H., Zhao, X., Gao, Y.: Meshnet: mesh neural network for 3d shape representation. In: Proceedings of the AAAI Conference on Artificial Intelligence 33, 8279\u20138286 (2019)","DOI":"10.1609\/aaai.v33i01.33018279"},{"issue":"4","key":"4453_CR44","first-page":"1","volume":"4","author":"O Sorkine","year":"2005","unstructured":"Sorkine, O.: Laplacian mesh processing. Eurographics (State of the Art Reports) 4(4), 1 (2005)","journal-title":"Eurographics (State of the Art Reports)"},{"key":"4453_CR45","unstructured":"Sorkine, O., Alexa, M.: As-rigid-as-possible surface modeling. In: Symposium on Geometry processing 4, 109\u2013116 (2007)"},{"key":"4453_CR46","doi-asserted-by":"crossref","unstructured":"Nath, U., Goel, R., Jeon, E.S., Kim, C., Min, K., Yang, Y., Yang, Y., Turaga, P.: Deep geometric moments promote shape consistency in text-to-3d generation. In: 2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 4331\u20134341. IEEE, (2025)","DOI":"10.1109\/WACV61041.2025.00425"},{"issue":"6","key":"4453_CR47","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3414685.3417861","volume":"39","author":"S Laine","year":"2020","unstructured":"Laine, S., Hellsten, J., Karras, T., Seol, Y., Lehtinen, J., Aila, T.: Modular primitives for high-performance differentiable rendering. ACM Trans. Graph. (ToG) 39(6), 1\u201314 (2020)","journal-title":"ACM Trans. Graph. (ToG)"},{"key":"4453_CR48","doi-asserted-by":"crossref","unstructured":"Kato, H., Ushiku, Y., Harada, T.: Neural 3d mesh renderer. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3907\u20133916 (2018)","DOI":"10.1109\/CVPR.2018.00411"},{"key":"4453_CR49","doi-asserted-by":"crossref","unstructured":"Cheng, Y.-C., Lee, H.-Y., Tulyakov, S., Schwing, A.G, Gui, L.-Y.: Sdfusion: multimodal 3d shape completion, reconstruction, and generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4456\u20134465 (2023)","DOI":"10.1109\/CVPR52729.2023.00433"},{"issue":"12","key":"4453_CR50","doi-asserted-by":"publisher","first-page":"7749","DOI":"10.1109\/TVCG.2024.3361502","volume":"30","author":"J Zhang","year":"2024","unstructured":"Zhang, J., Li, X., Wan, Z., Wang, C., Liao, J.: Text2nerf: text-driven 3d scene generation with neural radiance fields. IEEE Trans. Visual Comput. Graphics 30(12), 7749\u20137762 (2024)","journal-title":"IEEE Trans. Visual Comput. Graphics"},{"key":"4453_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, X., Yin, B.-W., Chen, Y., Lin, Z., Li, Y., Hou, Q., Cheng, M.-M.: Temo: towards text-driven 3d stylization for multi-object meshes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19531\u201319540 (2024)","DOI":"10.1109\/CVPR52733.2024.01847"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04453-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04453-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04453-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T13:38:42Z","timestamp":1775741922000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04453-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":51,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["4453"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04453-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]},"assertion":[{"value":"18 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"226"}}