{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:07Z","timestamp":1776931147892,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763977","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Training-Free Instance-Aware 3D Scene Reconstruction and Diffusion-Based View Synthesis from Sparse Images"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7504-5225","authenticated-orcid":false,"given":"Jiatong","family":"Xia","sequence":"first","affiliation":[{"name":"The University of Adelaide, Adelaide, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3584-795X","authenticated-orcid":false,"given":"Lingqiao","family":"Liu","sequence":"additional","affiliation":[{"name":"The University of Adelaide, Adelaide, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00580"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Jonathan\u00a0T. Barron Ben Mildenhall Dor Verbin Pratul\u00a0P. Srinivasan and Peter Hedman. 2023. Zip-NeRF: Anti-Aliased Grid-Based Neural Radiance Fields. ICCV (2023).","DOI":"10.1109\/ICCV51070.2023.01804"},{"key":"e_1_3_3_2_4_1","unstructured":"Shariq\u00a0Farooq Bhat Reiner Birkl Diana Wofk Peter Wonka and Matthias M\u00fcller. 2023. Zoedepth: Zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.12288 (2023)."},{"key":"e_1_3_3_2_5_1","unstructured":"Jiazhong Cen Zanwei Zhou Jiemin Fang Wei Shen Lingxi Xie Dongsheng Jiang Xiaopeng Zhang Qi Tian et\u00a0al. 2023. Segment anything in 3d with nerfs. Advances in Neural Information Processing Systems 36 (2023) 25971\u201325990."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19824-3_20"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00511"},{"key":"e_1_3_3_2_8_1","first-page":"370","volume-title":"ECCV","author":"Chen Yuedong","year":"2024","unstructured":"Yuedong Chen, Haofei Xu, Chuanxia Zheng, Bohan Zhuang, Marc Pollefeys, Andreas Geiger, Tat-Jen Cham, and Jianfei Cai. 2024b. Mvsplat: Efficient 3d gaussian splatting from sparse multi-view images. In ECCV. 370\u2013386."},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"crossref","unstructured":"Yaosen Chen Qi Yuan Zhiqiang Li Yuegen Liu Wei Wang Chaoping Xie Xuming Wen and Qien Yu. 2024c. Upst-nerf: Universal photorealistic style transfer of neural radiance fields for 3d scene. IEEE Transactions on Visualization and Computer Graphics (2024).","DOI":"10.1109\/TVCG.2024.3378692"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00127"},{"key":"e_1_3_3_2_11_1","volume-title":"ICML","author":"Cheng Kai","year":"2024","unstructured":"Kai Cheng, Xiaoxiao Long, Kaizhi Yang, Yao Yao, Wei Yin, Yuexin Ma, Wenping Wang, and Xuejin Chen. 2024. Gaussianpro: 3d gaussian splatting with progressive propagation. In ICML."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_3_2_13_1","first-page":"432","volume-title":"European Conference on Computer Vision","author":"Duan Yiquan","year":"2024","unstructured":"Yiquan Duan, Xianda Guo, and Zheng Zhu. 2024. Diffusiondepth: Diffusion denoising approach for monocular depth estimation. In European Conference on Computer Vision. Springer, 432\u2013449."},{"key":"e_1_3_3_2_14_1","unstructured":"Bardienus Duisterhof Lojze Zust Philippe Weinzaepfel Vincent Leroy Yohann Cabon and Jerome Revaud. 2024. MASt3R-SfM: a Fully-Integrated Solution for Unconstrained Structure-from-Motion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.19152 (2024)."},{"key":"e_1_3_3_2_15_1","first-page":"2366","volume-title":"NeurIPS","author":"Eigen David","year":"2014","unstructured":"David Eigen, Christian Puhrsch, and Rob Fergus. 2014. Depth map prediction from a single image using a multi-scale deep network. In NeurIPS. 2366\u20132374."},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687643"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01808"},{"key":"e_1_3_3_2_18_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. NeurIPS (2020)."},{"key":"e_1_3_3_2_19_1","unstructured":"Jonathan Ho Tim Salimans Alexey Gritsenko William Chan Mohammad Norouzi and David\u00a0J Fleet. 2022. Video diffusion models. NeurIPS (2022) 8633\u20138646."},{"key":"e_1_3_3_2_20_1","unstructured":"Jie Hu Shizun Wang and Xinchao Wang. 2025. PE3R: Perception-Efficient 3D Reconstruction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.07507 (2025)."},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657428"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02487"},{"key":"e_1_3_3_2_23_1","unstructured":"Hanwen Jiang Hao Tan Peng Wang Haian Jin Yue Zhao Sai Bi Kai Zhang Fujun Luan Kalyan Sunkavalli Qixing Huang and Georgios Pavlakos. 2025. RayZer: A Self-supervised Large View Synthesis Model. (2025)."},{"key":"e_1_3_3_2_24_1","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Jin Haian","year":"2025","unstructured":"Haian Jin, Hanwen Jiang, Hao Tan, Kai Zhang, Sai Bi, Tianyuan Zhang, Fujun Luan, Noah Snavely, and Zexiang Xu. 2025. LVSM: A Large View Synthesis Model with Minimal 3D Inductive Bias. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00907"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"crossref","unstructured":"Bernhard Kerbl Georgios Kopanas Thomas Leimk\u00fchler and George Drettakis. 2023. 3D Gaussian Splatting for Real-Time Radiance Field Rendering. ACM TOG 42 4 (July 2023).","DOI":"10.1145\/3592433"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01807"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01963"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00623"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3681758.3698002"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00194"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_24"},{"key":"e_1_3_3_2_34_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et\u00a0al. 2023. Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.07193 (2023)."},{"key":"e_1_3_3_2_35_1","unstructured":"Ben Poole Ajay Jain Jonathan\u00a0T Barron and Ben Mildenhall. 2022. Dreamfusion: Text-to-3d using 2d diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14988 (2022)."},{"key":"e_1_3_3_2_36_1","unstructured":"Nikhila Ravi Valentin Gabeur Yuan-Ting Hu Ronghang Hu Chaitanya Ryali Tengyu Ma Haitham Khedr Roman R\u00e4dle Chloe Rolland Laura Gustafson et\u00a0al. 2024. Sam 2: Segment anything in images and videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.00714 (2024)."},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.445"},{"key":"e_1_3_3_2_38_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2021. Denoising Diffusion Implicit Models. ICLR (2021)."},{"key":"e_1_3_3_2_39_1","unstructured":"Jiaxiang Tang Jiawei Ren Hang Zhou Ziwei Liu and Gang Zeng. 2024. Dreamgaussian: Generative gaussian splatting for efficient 3d content creation. ICLR (2024)."},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"crossref","unstructured":"Zhenggang Tang Yuchen Fan Dilin Wang Hongyu Xu Rakesh Ranjan Alexander Schwing and Zhicheng Yan. 2025. MV-DUSt3R+: Single-Stage Scene Reconstruction from Sparse Views In 2 Seconds. IEEE Conf. Comput. Vis. Pattern Recog. (2025).","DOI":"10.1109\/CVPR52734.2025.00498"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00381"},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"crossref","unstructured":"Qianqian Wang Yifei Zhang Aleksander Holynski Alexei\u00a0A Efros and Angjoo Kanazawa. 2025. Continuous 3D Perception Model with Persistent State. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12387 (2025).","DOI":"10.1109\/CVPR52734.2025.00983"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01956"},{"key":"e_1_3_3_2_44_1","first-page":"404","volume-title":"European Conference on Computer Vision","author":"Wang Yuxuan","year":"2024","unstructured":"Yuxuan Wang, Xuanyu Yi, Zike Wu, Na Zhao, Long Chen, and Hanwang Zhang. 2024b. View-consistent 3d editing with gaussian splatting. In European Conference on Computer Vision. Springer, 404\u2013420."},{"key":"e_1_3_3_2_45_1","first-page":"37","volume-title":"European Conference on Computer Vision","author":"Xu Tian-Xing","year":"2024","unstructured":"Tian-Xing Xu, Wenbo Hu, Yu-Kun Lai, Ying Shan, and Song-Hai Zhang. 2024. Texture-gs: Disentangling the geometry and texture for 3d gaussian splatting editing. In European Conference on Computer Vision. Springer, 37\u201353."},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"crossref","unstructured":"Jianing Yang Alexander Sax Kevin\u00a0J Liang Mikael Henaff Hao Tang Ang Cao Joyce Chai Franziska Meier and Matt Feiszli. 2025. Fast3R: Towards 3D Reconstruction of 1000+ Images in One Forward Pass. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.13928 (2025).","DOI":"10.1109\/CVPR52734.2025.02042"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"crossref","unstructured":"Lihe Yang Bingyi Kang Zilong Huang Zhen Zhao Xiaogang Xu Jiashi Feng and Hengshuang Zhao. 2024b. Depth anything v2. Advances in Neural Information Processing Systems 37 (2024) 21875\u201321911.","DOI":"10.52202\/079017-0688"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_47"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00567"},{"key":"e_1_3_3_2_51_1","volume-title":"European Conference on Computer Vision","author":"Ye Mingqiao","year":"2024","unstructured":"Mingqiao Ye, Martin Danelljan, Fisher Yu, and Lei Ke. 2024. Gaussian Grouping: Segment and Edit Anything in 3D Scenes. In European Conference on Computer Vision."},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00008"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02425"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01839"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00202"},{"key":"e_1_3_3_2_56_1","unstructured":"Chaoning Zhang Dongshen Han Sheng Zheng Jinwoo Choi Tae-Ho Kim and Choong\u00a0Seon Hong. 2023a. Mobilesamv2: Faster segment anything to everything. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.09579 (2023)."},{"key":"e_1_3_3_2_57_1","first-page":"335","volume-title":"European Conference on Computer Vision","author":"Zhang Jiawei","year":"2024","unstructured":"Jiawei Zhang, Jiahe Li, Xiaohan Yu, Lei Huang, Lin Gu, Jin Zheng, and Xiao Bai. 2024. Cor-gs: sparse-view 3d gaussian splatting via co-regularization. In European Conference on Computer Vision. 335\u2013352."},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"crossref","unstructured":"Lvmin Zhang Anyi Rao and Maneesh Agrawala. 2023b. Adding Conditional Control to Text-to-Image Diffusion Models.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_59_1","unstructured":"Chen Zhao Xuan Wang Tong Zhang Saqib Javed and Mathieu Salzmann. 2025. Self-Ensembling Gaussian Splatting for Few-Shot Novel View Synthesis. Proceedings of the IEEE\/CVF International Conference on Computer Vision (2025)."},{"key":"e_1_3_3_2_60_1","unstructured":"Jensen\u00a0(Jinghao) Zhou Hang Gao Vikram Voleti Aaryaman Vasishta Chun-Han Yao Mark Boss Philip Torr Christian Rupprecht and Varun Jampani. 2025. Stable Virtual Camera: Generative View Synthesis with Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.14489 (2025)."},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02048"},{"key":"e_1_3_3_2_62_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Ziwen Chen","year":"2025","unstructured":"Chen Ziwen, Hao Tan, Kai Zhang, Sai Bi, Fujun Luan, Yicong Hong, Li Fuxin, and Zexiang Xu. 2025. Long-LRM: Long-sequence Large Reconstruction Model for Wide-coverage Gaussian Splats. In Proceedings of the IEEE\/CVF International Conference on Computer Vision."}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763977","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:28:34Z","timestamp":1765250914000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763977"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":61,"alternative-id":["10.1145\/3757377.3763977","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763977","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}