{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:06:12Z","timestamp":1784268372813,"version":"3.55.0"},"reference-count":93,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100004344","name":"Adobe","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004344","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006785","name":"Google","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00468","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"4918-4929","source":"Crossref","is-referenced-by-count":3,"title":["Rayzer: a Self-Supervised Large View Synthesis Model"],"prefix":"10.1109","author":[{"given":"Hanwen","family":"Jiang","sequence":"first","affiliation":[{"name":"The University of Texas at Austin"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Tan","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Wang","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haian","family":"Jin","sequence":"additional","affiliation":[{"name":"Cornell University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Zhao","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sai","family":"Bi","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Zhang","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fujun","family":"Luan","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kalyan","family":"Sunkavalli","sequence":"additional","affiliation":[{"name":"Adobe Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qixing","family":"Huang","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Georgios","family":"Pavlakos","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/1468075.1468082"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2102.05095"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00405"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00027"},{"key":"ref6","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"NeurIPS"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00574"},{"key":"ref8","first-page":"16123","article-title":"Ef cient geometry-aware 3d generative adversarial networks","volume-title":"CVPR","author":"Chan","year":"2022"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01840"},{"key":"ref10","first-page":"14124","article-title":"Mvsnerf: Fast general izable radiance eld reconstruction from multi-view stereo","volume-title":"ICCV","author":"Chen","year":"2021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00653"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3598711\/mm2"},{"key":"ref13","first-page":"370","article-title":"Mvsplat: Ef cient 3d gaussian splatting from sparse multi-view images","volume-title":"ECCV","author":"Chen","year":"2024"},{"key":"ref14","first-page":"628","article-title":"3d-r2n2: A uni ed approach for single and multi-view 3d object reconstruction","volume-title":"ECCV","author":"Choy","year":"2016"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01263"},{"key":"ref16","author":"Dosovitskiy","year":"2020","journal-title":"An image is worth 16x16 words: Trans formers for image recognition at scale"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00481"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/3dv66043.2025.00008"},{"key":"ref19","first-page":"10392","article-title":"Mononerf: Learning generalizable nerfs from monocular videos without cam era poses","volume-title":"International Conference on Machine Learning","author":"Fu","year":"2023"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01965"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_29"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr42600.2020.00256"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00780"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-60243-6_11"},{"key":"ref25","author":"Hong","year":"2023","journal-title":"Lrm: Large reconstruction model for single image to 3d"},{"key":"ref26","author":"Jiang","year":"2023","journal-title":"Leap: Liberate sparse-view 3d modeling from camera poses"},{"key":"ref27","author":"Jiang","year":"2024","journal-title":"Real3d: Scaling up large reconstruction models with real world images"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00055"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.01533"},{"key":"ref30","volume-title":"Lvsm: A large view synthesis model with minimal 3d inductive bias","author":"Jin","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.00982"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1603.08155"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/15922.15902"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_23"},{"key":"ref35","author":"Kaplan","year":"2020","journal-title":"Scaling laws for neural language models"},{"key":"ref36","article-title":"Learning a multi-view stereo machine","volume":"30","author":"Kar","year":"2017","journal-title":"NIPS"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3592433"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00959"},{"key":"ref39","first-page":"71","article-title":"Grounding image matching in 3d with mast3r","author":"Leroy","year":"2024","journal-title":"ECCV"},{"key":"ref40","first-page":"22554","article-title":"An analysis of svd for deep rotation estimation","volume":"33","author":"Levinson","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00218"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_11"},{"key":"ref43","first-page":"11453","article-title":"Sdf-srn: Learning signed distance 3d object reconstruction from static images","volume":"33","author":"Lin","year":"2020","journal-title":"NeurIPS"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00569"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02092"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/3503250"},{"key":"ref47","first-page":"3971","article-title":"Selfsupervised viewpoint learning from image collections","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Mustikovela"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00255"},{"key":"ref49","author":"Nichol","year":"2022","journal-title":"Point-e: A system for generating 3d point clouds from complex prompts"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00025"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1098\/rstl.1865.0017"},{"key":"ref53","first-page":"5648","article-title":"Volumetric and multi-view cnns for object classification on 3d data","author":"Charles R","year":"2016","journal-title":"CVPR"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00102"},{"issue":"8","key":"ref55","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"ref56","article-title":"Learning internal representations by error propagation","author":"David E","year":"1985","journal-title":"Learning internal representations by error propagation"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00613"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01659"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00391"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.445"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/3dv66043.2025.00041"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1177\/089443938600400104"},{"key":"ref63","article-title":"Lecture 15: Structure from motion","author":"Snavely","year":"2017","journal-title":"Lecture 15: Structure from motion"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00408"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00039"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2019.00270"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73397-0_12"},{"key":"ref69","author":"Wang","year":"2023","journal-title":"Pf-lrm: Pose-free large reconstruction model for joint pose and shape prediction"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00466"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00983"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01956"},{"key":"ref73","author":"Wei","year":"2024","journal-title":"Meshlrm: Large reconstruction model for highquality meshes"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0253"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00875"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00226"},{"key":"ref77","article-title":"Perspective transformer nets: Learning singleview 3d object reconstruction without 3d supervision","author":"Yan","year":"2016","journal-title":"NIPS"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02042"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"ref80","first-page":"21875","article-title":"Depth anything v2","volume":"37","author":"Yang","year":"2025","journal-title":"NeurIPS"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00455"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3259681"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1145\/3592442"},{"key":"ref84","first-page":"592","article-title":"Relpose: Predicting probabilistic relative rotation for single objects in the wild","author":"Jason Y","year":"2022","journal-title":"ECCV"},{"key":"ref85","author":"Jason Y","year":"2024","journal-title":"Cameras as rays: Pose estimation via ray diffusion"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72670-5_1"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02043"},{"key":"ref88","article-title":"Relitlrm: Generative relightable radiance for large reconstruction models","author":"Zhang","year":"2024","journal-title":"Relitlrm: Generative relightable radiance for large reconstruction models"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_2"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.700"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201323"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00589"},{"key":"ref93","author":"Chen","year":"2024","journal-title":"Long-lrm: Long-sequence large reconstruction model for wide-coverage gaussian splats"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445952.pdf?arnumber=11445952","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:33:11Z","timestamp":1777613591000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445952\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":93,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00468","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}