{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:12:41Z","timestamp":1784268761084,"version":"3.55.0"},"reference-count":101,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,9,1]],"date-time":"2025-09-01T00:00:00Z","timestamp":1756684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61977047"],"award-info":[{"award-number":["61977047"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976138"],"award-info":[{"award-number":["61976138"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["21010502400"],"award-info":[{"award-number":["21010502400"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"name":"MoE Key Lab of Intelligent Perception and Human-Machine Collaboration"},{"name":"Shanghai Frontiers Science Center of Human-centered Artificial Intelligence"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1109\/tpami.2024.3463875","type":"journal-article","created":{"date-parts":[[2024,9,18]],"date-time":"2024-09-18T13:56:25Z","timestamp":1726667785000},"page":"7206-7217","source":"Crossref","is-referenced-by-count":7,"title":["HOLI-1-to-3: Transient-Enhanced Holistic Image-to-3D Generation"],"prefix":"10.1109","volume":"47","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2922-1718","authenticated-orcid":false,"given":"Siyuan","family":"Shen","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1779-4217","authenticated-orcid":false,"given":"Suan","family":"Xia","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2198-8279","authenticated-orcid":false,"given":"Xingyue","family":"Peng","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4697-5183","authenticated-orcid":false,"given":"Ziyu","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingsheng","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4222-0648","authenticated-orcid":false,"given":"Shiying","family":"Li","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9198-6853","authenticated-orcid":false,"given":"Jingyi","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, ShanghaiTech University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5540082"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00539"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000025798.50602.3a"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.381"},{"key":"ref5","first-page":"27","article-title":"Multi-platform document-oriented GUIs","volume-title":"Proc. 10th Australas. Conf. User Interfaces","author":"Kim"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00420"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20071-7_33"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"ref9","article-title":"Magic123: One image to high-quality 3D object generation using both 2D and 3D diffusion priors","author":"Qian","year":"2023"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1038\/ncomms1747"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1364\/OE.20.019096"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1364\/OE.23.020997"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1038\/nature25489"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3306346.3322937"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1461-3"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1038\/nphoton.2015.234"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1364\/OE.25.010109"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1364\/OE.455803"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1364\/OE.25.017466"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCP54855.2022.9887660"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3076062"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00696"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00164"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00148"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3203383"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01279"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01565"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3181070"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00816"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"ref31","article-title":"An image is worth one word: Personalizing text-to-image generation using textual inversion","author":"Gal","year":"2022"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref33","first-page":"6087","article-title":"Deep marching tetrahedra: A hybrid representation for high-resolution 3D shape synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shen"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01263"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01534"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5540088"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2000.854854"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1038\/s41579-020-00440-4"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00651"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCP51581.2021.9466270"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/2461912.2461928"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1364\/OE.439372"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-020-15157-4"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3414685.3417825"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1364\/OE.443127"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00233"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3102163.3102241"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00966"},{"key":"ref49","article-title":"Omni-line-of-sight imaging for holistic shape reconstruction","author":"Huang","year":"2023"},{"key":"ref50","first-page":"540","article-title":"MarrNet: 3D shape reconstruction via 2.5D sketches","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_40"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_4"},{"key":"ref53","first-page":"5866","article-title":"GEOMetrics: Exploiting geometric structure for graph-encoded objects","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Smith"},{"key":"ref54","first-page":"2807","article-title":"Unsupervised learning of shape and pose with differentiable point clouds","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Insafutdinov"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2018.00030"},{"key":"ref56","first-page":"2107","article-title":"Octree generating networks: Efficient convolutional architectures for high-resolution 3D outputs","volume-title":"Proc. IEEE Int. Conf. Comput. Vis.","author":"Tatarchenko"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/3272127.3275050"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00459"},{"key":"ref59","first-page":"490","article-title":"DISN: Deep implicit surface network for high-quality single-view 3D reconstruction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00314"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00352"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01041"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14689"},{"key":"ref64","article-title":"StyleNeRF: A style-based 3D aware generator for high-resolution image synthesis","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gu"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00431"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00430"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/3544777"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00825"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459838"},{"key":"ref70","article-title":"3D generation on imagenet","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Skorokhodov"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00391"},{"key":"ref72","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref73","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Saharia"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00286"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00577"},{"key":"ref76","first-page":"10021","article-title":"Lion: Latent point diffusion models for 3D shape generation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zeng"},{"key":"ref77","article-title":"Point-E: A system for generating 3D point clouds from complex prompts","author":"Nichol","year":"2022"},{"key":"ref78","article-title":"3DGen: Triplane latent diffusion for textured mesh generation","author":"Gupta","year":"2023"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1145\/3592442"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1145\/3592103"},{"key":"ref81","first-page":"14254","article-title":"HyperDiffusion: Generating implicit neural fields with weight-space diffusion","volume-title":"Proc. Int. Conf. Comput. Vis.","author":"Erko\u00e7"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02000"},{"key":"ref83","article-title":"Michelangelo: Conditional 3D shape generation based on shape-image-text aligned latent representation","author":"Zhao","year":"2023"},{"key":"ref84","article-title":"Shap-E: Generating conditional 3D implicit functions","author":"Jun","year":"2023"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00421"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02100"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00229"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-022-1411-7"},{"key":"ref89","article-title":"DreamFusion: Text-to-3D using 2D diffusion","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Poole"},{"key":"ref90","first-page":"22226","article-title":"One-2\u20133-45: Any single image to 3D mesh in 45 seconds without per-shape optimization","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Liu"},{"key":"ref91","article-title":"SyncDreamer: Learning to generate multiview-consistent images from a single-view image","author":"Liu","year":"2023"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.00676"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.00951"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00747"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00435"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/JPHOT.2022.3207785"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1145\/3528223.3530127"},{"key":"ref98","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sohl-Dickstein"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01263"},{"key":"ref100","article-title":"Threestudio: A unified framework for 3D content generation","author":"Guo","year":"2023"},{"key":"ref101","article-title":"ProlificDreamer: High-fidelity and diverse text-to-3D generation with variational score distillation","author":"Wang","year":"2023"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11118328\/10684158.pdf?arnumber=10684158","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,7]],"date-time":"2025-08-07T17:44:24Z","timestamp":1754588664000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10684158\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9]]},"references-count":101,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3463875","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9]]}}}