{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:08:22Z","timestamp":1784268502126,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","funder":[{"name":"ERC Consolidator Grant Gen3D","award":["101171131"],"award-info":[{"award-number":["101171131"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763946","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["WorldExplorer: Towards Generating Fully Navigable 3D Scenes"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6778-1191","authenticated-orcid":false,"given":"Manuel-Andreas","family":"Schneider","sequence":"first","affiliation":[{"name":"Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9494-0797","authenticated-orcid":false,"given":"Lukas","family":"H\u00f6llein","sequence":"additional","affiliation":[{"name":"Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6093-5199","authenticated-orcid":false,"given":"Matthias","family":"Nie\u00dfner","sequence":"additional","affiliation":[{"name":"Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02130"},{"key":"e_1_3_3_2_3_1","unstructured":"Sherwin Bahmani Ivan Skorokhodov Aliaksandr Siarohin Willi Menapace Guocheng Qian Michael Vasilkovsky Hsin-Ying Lee Chaoyang Wang Jiaxu Zou Andrea Tagliasacchi et\u00a0al. 2024. Vd3d: Taming large video diffusion transformers for 3d camera control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.12781 (2024)."},{"key":"e_1_3_3_2_4_1","unstructured":"Jianhong Bai Menghan Xia Xiao Fu Xintao Wang Lianrui Mu Jinwen Cao Zuozhu Liu Haoji Hu Xiang Bai Pengfei Wan et\u00a0al. 2025. ReCamMaster: Camera-Controlled Generative Rendering from A Single Video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.11647 (2025)."},{"key":"e_1_3_3_2_5_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et\u00a0al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.15127 (2023)."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_3_2_7_1","unstructured":"Luxi Chen Zihan Zhou Min Zhao Yikai Wang Ge Zhang Wenhao Huang Hao Sun Ji-Rong Wen and Chongxuan Li. 2025b. FlexWorld: Progressively Expanding 3D Scenes for Flexiable-View Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.13265 (2025)."},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"crossref","unstructured":"Sili Chen Hengkai Guo Shengnan Zhu Feihu Zhang Zilong Huang Jiashi Feng and Bingyi Kang. 2025a. Video Depth Anything: Consistent Depth Estimation for Super-Long Videos. arXiv:https:\/\/arXiv.org\/abs\/2501.12375 (2025).","DOI":"10.1109\/CVPR52734.2025.02126"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00314"},{"key":"e_1_3_3_2_11_1","unstructured":"Ruiqi Gao Aleksander Holynski Philipp Henzler Arthur Brussee Ricardo Martin-Brualla Pratul\u00a0P. Srinivasan Jonathan\u00a0T. Barron and Ben Poole. 2024. CAT3D: Create Anything in 3D with Multi-View Diffusion Models. Advances in Neural Information Processing Systems (2024)."},{"key":"e_1_3_3_2_12_1","first-page":"20061","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Goli Lily","year":"2024","unstructured":"Lily Goli, Cody Reading, Silvia Sell\u00e1n, Alec Jacobson, and Andrea Tagliasacchi. 2024. Bayes\u2019 rays: Uncertainty quantification for neural radiance fields. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 20061\u201320070."},{"key":"e_1_3_3_2_13_1","unstructured":"Hao He Ceyuan Yang Shanchuan Lin Yinghao Xu Meng Wei Liangke Gui Qi Zhao Gordon Wetzstein Lu Jiang and Hongsheng Li. 2025. CameraCtrl II: Dynamic Scene Exploration via Camera-controlled Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.10592 (2025)."},{"key":"e_1_3_3_2_14_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020) 6840\u20136851."},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00482"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00727"},{"key":"e_1_3_3_2_17_1","volume-title":"The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022","author":"Hu Edward\u00a0J.","year":"2022","unstructured":"Edward\u00a0J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In The Tenth International Conference on Learning Representations, ICLR 2022, Virtual Event, April 25-29, 2022. OpenReview.net. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00907"},{"key":"e_1_3_3_2_19_1","unstructured":"Bingxin Ke Kevin Qu Tianfu Wang Nando Metzger Shengyu Huang Bo Li Anton Obukhov and Konrad Schindler. 2025. Marigold: Affordable Adaptation of Diffusion-Based Image Generators for Image Analysis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.09358 (2025)."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"crossref","unstructured":"Bernhard Kerbl Georgios Kopanas Thomas Leimk\u00fchler and George Drettakis. 2023. 3d gaussian splatting for real-time radiance field rendering. ACM Trans. Graph. 42 4 (2023) 139\u20131.","DOI":"10.1145\/3592433"},{"key":"e_1_3_3_2_21_1","unstructured":"Peter Kocsis Lukas H\u00f6llein and Matthias Nie\u00dfner. 2025. IntrinsiX: High-Quality PBR Generation using Image Priors. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.01008 (2025)."},{"key":"e_1_3_3_2_22_1","unstructured":"Black\u00a0Forest Labs. 2023. FLUX. https:\/\/github.com\/black-forest-labs\/flux."},{"key":"e_1_3_3_2_23_1","first-page":"214","volume-title":"European Conference on Computer Vision","author":"Li Haoran","year":"2024","unstructured":"Haoran Li, Haolin Shi, Wenli Zhang, Wenjun Wu, Yong Liao, Lin Wang, Lik-hang Lee, and Peng\u00a0Yuan Zhou. 2024b. Dreamscene: 3d gaussian-based text-to-3d scene generation via formation pattern sampling. In European Conference on Computer Vision. Springer, 214\u2013230."},{"key":"e_1_3_3_2_24_1","unstructured":"Wenrui Li Fucheng Cai Yapeng Mi Zhe Yang Wangmeng Zuo Xingtao Wang and Xiaopeng Fan. 2024a. Scenedreamer360: Text-driven 3d-consistent scene generation with panoramic gaussian splatting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.13711 (2024)."},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"crossref","unstructured":"Hanwen Liang Junli Cao Vidit Goel Guocheng Qian Sergei Korolev Demetri Terzopoulos Konstantinos\u00a0N Plataniotis Sergey Tulyakov and Jian Ren. 2024. Wonderland: Navigating 3D Scenes from a Single Image. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.12091 (2024).","DOI":"10.1109\/CVPR52734.2025.00083"},{"key":"e_1_3_3_2_26_1","unstructured":"Yaron Lipman Ricky\u00a0TQ Chen Heli Ben-Hamu Maximilian Nickel and Matt Le. 2022. Flow matching for generative modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.02747 (2022)."},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01419"},{"key":"e_1_3_3_2_28_1","unstructured":"Fangfu Liu Wenqiang Sun Hanyang Wang Yikai Wang Haowen Sun Junliang Ye Jun Zhang and Yueqi Duan. 2024b. ReconX: Reconstruct Any Scene from Sparse Views with Video Diffusion Model. arxiv:https:\/\/arXiv.org\/abs\/2408.16767\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2408.16767"},{"key":"e_1_3_3_2_29_1","unstructured":"Kunhao Liu Ling Shao and Shijian Lu. 2024a. Novel View Extrapolation with Video Diffusion Priors. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.14208 (2024)."},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","unstructured":"Dustin Podell Zion English Kyle Lacey Andreas Blattmann Tim Dockhorn Jonas M\u00fcller Joe Penna and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. CoRR abs\/2307.01952 (2023). 10.48550\/ARXIV.2307.01952 arXiv:https:\/\/arXiv.org\/abs\/2307.01952","DOI":"10.48550\/ARXIV.2307.01952"},{"key":"e_1_3_3_2_34_1","volume-title":"The Eleventh International Conference on Learning Representations, ICLR 2023, Kigali, Rwanda, May 1-5, 2023","author":"Poole Ben","year":"2023","unstructured":"Ben Poole, Ajay Jain, Jonathan\u00a0T. Barron, and Ben Mildenhall. 2023. DreamFusion: Text-to-3D using 2D Diffusion. In The Eleventh International Conference on Learning Representations, ICLR 2023, Kigali, Rwanda, May 1-5, 2023. OpenReview.net. https:\/\/openreview.net\/forum?id=FjNys5c7VyY"},{"key":"e_1_3_3_2_35_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_36_1","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06125 1 2 (2022) 3."},{"key":"e_1_3_3_2_37_1","unstructured":"Nikhila Ravi Jeremy Reizenstein David Novotny Taylor Gordon Wan-Yen Lo Justin Johnson and Georgia Gkioxari. 2020. Accelerating 3d deep learning with pytorch3d. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2007.08501 (2020)."},{"key":"e_1_3_3_2_38_1","unstructured":"Xuanchi Ren Tianchang Shen Jiahui Huang Huan Ling Yifan Lu Merlin Nimier-David Thomas M\u00fcller Alexander Keller Sanja Fidler and Jun Gao. 2025. Gen3c: 3d-informed world-consistent video generation with precise camera control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.03751 (2025)."},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01976"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"crossref","unstructured":"Chitwan Saharia William Chan Saurabh Saxena Lala Li Jay Whang Emily Denton Seyed Kamyar\u00a0Seyed Ghasemipour Burcu\u00a0Karagol Ayan S.\u00a0Sara Mahdavi Rapha\u00a0Gontijo Lopes Tim Salimans Jonathan Ho David\u00a0J Fleet and Mohammad Norouzi. 2022. Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding.","DOI":"10.1145\/3528233.3530757"},{"key":"e_1_3_3_2_44_1","unstructured":"Tim Salimans Ian Goodfellow Wojciech Zaremba Vicki Cheung Alec Radford and Xi Chen. 2016. Improved techniques for training gans. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.445"},{"key":"e_1_3_3_2_46_1","unstructured":"Christoph Schuhmann Romain Beaumont Richard Vencu Cade Gordon Ross Wightman Mehdi Cherti Theo Coombes Aarush Katta Clayton Mullis Mitchell Wortsman et\u00a0al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems 35 (2022) 25278\u201325294."},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00593"},{"key":"e_1_3_3_2_48_1","unstructured":"Katja Schwarz Denys Rozumnyi Samuel\u00a0Rota Bul\u00f2 Lorenzo Porzi and Peter Kontschieder. 2025. A Recipe for Generating 3D Worlds From a Single Image. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.16611 (2025)."},{"key":"e_1_3_3_2_49_1","unstructured":"Team Seawead Ceyuan Yang Zhijie Lin Yang Zhao Shanchuan Lin Zhibei Ma Haoyuan Guo Hao Chen Lu Qi Sen Wang et\u00a0al. 2025. Seaweed-7b: Cost-effective training of video generation foundation model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.08685 (2025)."},{"key":"e_1_3_3_2_50_1","unstructured":"Jaidev Shriram Alex Trevithick Lingjie Liu and Ravi Ramamoorthi. 2024. Realmdreamer: Text-driven 3d scene generation with inpainting and depth diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.07199 (2024)."},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","unstructured":"Yawar Siddiqui Tom Monnier Filippos Kokkinos Mahendra Kariya Yanir Kleiman Emilien Garreau Oran Gafni Natalia Neverova Andrea Vedaldi Roman Shapovalov and David Novotn\u00fd. 2024. Meta 3D AssetGen: Text-to-Mesh Generation with High-Quality Geometry Texture and PBR Materials. CoRR abs\/2407.02445 (2024). 10.48550\/ARXIV.2407.02445 arXiv:https:\/\/arXiv.org\/abs\/2407.02445","DOI":"10.48550\/ARXIV.2407.02445"},{"key":"e_1_3_3_2_52_1","unstructured":"Kiwhan Song Boyuan Chen Max Simchowitz Yilun Du Russ Tedrake and Vincent Sitzmann. 2025. History-Guided Video Diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.06764 (2025)."},{"key":"e_1_3_3_2_53_1","unstructured":"Wenqiang Sun Shuo Chen Fangfu Liu Zilong Chen Yueqi Duan Jun Zhang and Yikai Wang. 2024. Dimensionx: Create any 3d and 4d scenes from a single image with controllable video diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.04928 (2024)."},{"key":"e_1_3_3_2_54_1","unstructured":"Stanislaw Szymanowicz Jason\u00a0Y. Zhang Pratul Srinivasan Ruiqi Gao Arthur Brussee Aleksander Holynski Ricardo Martin-Brualla Jonathan\u00a0T. Barron and Philipp Henzler. 2025. Bolt3D: Generating 3D Scenes in Seconds. arXiv:https:\/\/arXiv.org\/abs\/2503.14445 (2025)."},{"key":"e_1_3_3_2_55_1","unstructured":"Shitao Tang Fuyang Zhang Jiacheng Chen Peng Wang and Yasutaka Furukawa. 2023. MVDiffusion: Enabling Holistic Multi-view Image Generation with Correspondence-Aware Diffusion. arXiv (2023)."},{"key":"e_1_3_3_2_56_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"crossref","unstructured":"Guangcong Wang Peng Wang Zhaoxi Chen Wenping Wang Chen\u00a0Change Loy and Ziwei Liu. 2024. Perf: Panoramic neural radiance field from a single panorama. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024).","DOI":"10.1109\/TPAMI.2024.3387307"},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"crossref","unstructured":"Hanyang Wang Fangfu Liu Jiawei Chi and Yueqi Duan. 2025b. VideoScene: Distilling Video Diffusion Model to Generate 3D Scenes in One Step. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.01956 (2025).","DOI":"10.1109\/CVPR52734.2025.01536"},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"crossref","unstructured":"Jianyuan Wang Minghao Chen Nikita Karaev Andrea Vedaldi Christian Rupprecht and David Novotny. 2025a. Vggt: Visual geometry grounded transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.11651 (2025).","DOI":"10.1109\/CVPR52734.2025.00499"},{"key":"e_1_3_3_2_60_1","unstructured":"Zhengyi Wang Cheng Lu Yikai Wang Fan Bao Chongxuan Li Hang Su and Jun Zhu. 2023. Prolificdreamer: High-fidelity and diverse text-to-3d generation with variational score distillation. Advances in Neural Information Processing Systems 36 (2023) 8406\u20138441."},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00749"},{"key":"e_1_3_3_2_62_1","unstructured":"Enze Xie Junsong Chen Junyu Chen Han Cai Haotian Tang Yujun Lin Zhekai Zhang Muyang Li Ligeng Zhu Yao Lu et\u00a0al. 2024. Sana: Efficient high-resolution image synthesis with linear diffusion transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.10629 (2024)."},{"key":"e_1_3_3_2_63_1","doi-asserted-by":"crossref","unstructured":"Lihe Yang Bingyi Kang Zilong Huang Zhen Zhao Xiaogang Xu Jiashi Feng and Hengshuang Zhao. 2024a. Depth anything v2. Advances in Neural Information Processing Systems 37 (2024) 21875\u201321911.","DOI":"10.52202\/079017-0688"},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"crossref","unstructured":"Shuai Yang Jing Tan Mengchen Zhang Tong Wu Yixuan Li Gordon Wetzstein Ziwei Liu and Dahua Lin. 2024b. Layerpano3d: Layered 3d panorama for hyper-immersive scene generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.13252 (2024).","DOI":"10.1145\/3721238.3730643"},{"key":"e_1_3_3_2_65_1","unstructured":"Zhuoyi Yang Jiayan Teng Wendi Zheng Ming Ding Shiyu Huang Jiazheng Xu Yuanming Yang Wenyi Hong Xiaohan Zhang Guanyu Feng et\u00a0al. 2024c. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.06072 (2024)."},{"key":"e_1_3_3_2_66_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06721 (2023)."},{"key":"e_1_3_3_2_67_1","unstructured":"Hong-Xing Yu Haoyi Duan Charles Herrmann William\u00a0T Freeman and Jiajun Wu. 2024a. Wonderworld: Interactive 3d scene generation from a single image. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.09394 (2024)."},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00636"},{"key":"e_1_3_3_2_69_1","unstructured":"Wangbo Yu Jinbo Xing Li Yuan Wenbo Hu Xiaoyu Li Zhipeng Huang Xiangjun Gao Tien-Tsin Wong Ying Shan and Yonghong Tian. 2024c. Viewcrafter: Taming video diffusion models for high-fidelity novel view synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.02048 (2024)."},{"key":"e_1_3_3_2_70_1","unstructured":"Chenshuang Zhang Chaoning Zhang Mengchun Zhang and In\u00a0So Kweon. 2023b. Text-to-image diffusion models in generative ai: A survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.07909 (2023)."},{"key":"e_1_3_3_2_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00652"},{"key":"e_1_3_3_2_73_1","doi-asserted-by":"crossref","unstructured":"Shengjun Zhang Jinzhao Li Xin Fei Hao Liu and Yueqi Duan. 2025. Scene Splatter: Momentum 3D Scene Generation from Single Image with Video Diffusion Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.02764 (2025).","DOI":"10.1109\/CVPR52734.2025.00571"},{"key":"e_1_3_3_2_74_1","unstructured":"Jensen\u00a0Jinghao Zhou Hang Gao Vikram Voleti Aaryaman Vasishta Chun-Han Yao Mark Boss Philip Torr Christian Rupprecht and Varun Jampani. 2025. STABLE VIRTUAL CAMERA: Generative View Synthesis with Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.14489 (2025)."},{"key":"e_1_3_3_2_75_1","first-page":"324","volume-title":"European Conference on Computer Vision","author":"Zhou Shijie","year":"2024","unstructured":"Shijie Zhou, Zhiwen Fan, Dejia Xu, Haoran Chang, Pradyumna Chari, Tejas Bharadwaj, Suya You, Zhangyang Wang, and Achuta Kadambi. 2024. Dreamscene360: Unconstrained text-to-3d scene generation with panoramic gaussian splatting. In European Conference on Computer Vision. Springer, 324\u2013342."}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763946","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:26:12Z","timestamp":1765250772000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763946"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":74,"alternative-id":["10.1145\/3757377.3763946","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763946","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}