{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T10:02:13Z","timestamp":1777888933917,"version":"3.51.4"},"reference-count":61,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62422606,62201484"],"award-info":[{"award-number":["62422606,62201484"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00664","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"7069-7078","source":"Crossref","is-referenced-by-count":0,"title":["StableDepth: Scene-Consistent and Scale-Invariant Monocular Depth"],"prefix":"10.1109","author":[{"given":"Zheng","family":"Zhang","sequence":"first","affiliation":[{"name":"The University of Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lihe","family":"Yang","sequence":"additional","affiliation":[{"name":"The University of Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianyu","family":"Yang","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaohui","family":"Yu","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyang","family":"Guo","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yixing","family":"Lao","sequence":"additional","affiliation":[{"name":"The University of Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengshuang","family":"Zhao","sequence":"additional","affiliation":[{"name":"The University of Hong Kong"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Zoedepth: Zero-shot transfer by combining relative and metric depth.","author":"Bhat","year":"2023"},{"key":"ref2","volume-title":"Midas v3. 1-a model zoo for robust monocular relative depth estimation.","author":"Birk","year":"2023"},{"key":"ref3","volume-title":"Stable video diffusion: Scaling latent video diffusion models to large datasets.","author":"Blattmann","year":"2023"},{"key":"ref4","article-title":"Depth pro: Sharp monocular metric depth in less than a second","volume-title":"ICLR","author":"Bochkovskii","year":"2025"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33783-3_44"},{"key":"ref6","volume-title":"Virtual kitti 2","author":"Cabon","year":"2020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02126"},{"key":"ref8","article-title":"Singleimage depth perception in the wild","volume-title":"NeurIPS","author":"Chen","year":"2016"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"ref10","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","volume-title":"ICLR","author":"Dosovitskiy","year":"2021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01061"},{"key":"ref12","article-title":"Depth map prediction from a single image using a multi-scale deep network","volume-title":"NeurIPS","author":"Eigen","year":"2014"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1049\/iet-ipr.2018.5920"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45103-X_50"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72670-5_14"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2016.470"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913491297"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i3.32330"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00847"},{"key":"ref20","article-title":"Scale-invariant monocular depth estimation via ssi depth","volume-title":"SIGGRAPH","author":"Mahdi","year":"2024"},{"key":"ref21","article-title":"Video diffusion models","volume-title":"NeurIPS","author":"Ho","year":"2022"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3444912"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00193"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01768"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00907"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00678"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00166"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00297"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00303"},{"key":"ref30","volume-title":"Lightwheelocc: A 3d occupancy synthetic dataset in autonomous driving.","year":"2024"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2021.3122139"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/tmi.2019.2950936"},{"key":"ref33","article-title":"Decoupled weight decay regularization","volume-title":"ICLR","author":"Loshchilov","year":"2019"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392377"},{"key":"ref35","article-title":"Dinov2: Learning robust visual features without supervision","volume-title":"TMLR","author":"Oquab","year":"2023"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967590"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00963"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3628473"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2020.3019967"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1163\/2405-8262_rgg4_sim_025185"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01073"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW54805.2022.00072"},{"key":"ref45","article-title":"Make3d: Learning 3d scene structure from a single still image","volume-title":"TPAMI","author":"Saxena","year":"2008"},{"key":"ref46","article-title":"Simplere-con: 3d reconstruction without 3d convolutions","volume-title":"E C C V","author":"Sayed","year":"2022"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02127"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58529-7_34"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2019.00046"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01249"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00864"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00868"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00874"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8794182"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00069"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"ref57","article-title":"Depth anything v2","volume-title":"NeurIPS","author":"Yang","year":"2024"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00804"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00830"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19839-7_9"},{"key":"ref61","volume-title":"Scaledepth: Decomposing metric depth estimation into scale prediction and relative depth estimation.","author":"Zhu","year":"2024"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445848.pdf?arnumber=11445848","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:02:50Z","timestamp":1777611770000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445848\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":61,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00664","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}