{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:04:45Z","timestamp":1784228685951,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62422606"],"award-info":[{"award-number":["62422606"]}]},{"name":"Hong Kong Research Grant Council General Research Fund","award":["17213925"],"award-info":[{"award-number":["17213925"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811149","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SCOPE: Scale-Consistent One-Pass Estimation of 3D Geometry"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-5282-3661","authenticated-orcid":false,"given":"Zheng","family":"Zhang","sequence":"first","affiliation":[{"name":"The University of Hong Kong (HKU), Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8600-9733","authenticated-orcid":false,"given":"Lihe","family":"Yang","sequence":"additional","affiliation":[{"name":"The University of Hong Kong (HKU), Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9674-5220","authenticated-orcid":false,"given":"Tianyu","family":"Yang","sequence":"additional","affiliation":[{"name":"Meituan, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7852-4491","authenticated-orcid":false,"given":"Chaohui","family":"Yu","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8338-3577","authenticated-orcid":false,"given":"Yixing","family":"Lao","sequence":"additional","affiliation":[{"name":"The University of Hong Kong (HKU), Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8265-7441","authenticated-orcid":false,"given":"Xiaoyang","family":"Guo","sequence":"additional","affiliation":[{"name":"Horizon Robotics, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6156-0816","authenticated-orcid":false,"given":"Biao","family":"Gong","sequence":"additional","affiliation":[{"name":"Ant Group, HangZhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7320-1119","authenticated-orcid":false,"given":"Fan","family":"Wang","sequence":"additional","affiliation":[{"name":"Alibaba Group, HangZhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8277-2706","authenticated-orcid":false,"given":"Hengshuang","family":"Zhao","sequence":"additional","affiliation":[{"name":"The University of Hong Kong (HKU), Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","volume-title":"CVPR","author":"Bhat Shariq\u00a0Farooq","year":"2021","unstructured":"Shariq\u00a0Farooq Bhat, Ibraheem Alhashim, and Peter Wonka. 2021. Adabins: Depth estimation using adaptive bins. In CVPR."},{"key":"e_1_3_3_2_3_1","unstructured":"Shariq\u00a0Farooq Bhat Reiner Birkl Diana Wofk Peter Wonka and Matthias M\u00fcller. 2023. Zoedepth: Zero-shot transfer by combining relative and metric depth. arXiv:https:\/\/arXiv.org\/abs\/2302.12288 (2023)."},{"key":"e_1_3_3_2_4_1","unstructured":"Reiner Birkl Diana Wofk and Matthias M\u00fcller. 2023. MiDaS v3. 1\u2013A Model Zoo for Robust Monocular Relative Depth Estimation. arXiv:https:\/\/arXiv.org\/abs\/2307.14460 (2023)."},{"key":"e_1_3_3_2_5_1","volume-title":"ICLR","author":"Bochkovskii Aleksei","year":"2025","unstructured":"Aleksei Bochkovskii, Ama\u00ebl Delaunoy, Hugo Germain, Marcel Santos, Yichao Zhou, Stephan\u00a0R. Richter, and Vladlen Koltun. 2025. Depth Pro: Sharp Monocular Metric Depth in Less Than a Second. In ICLR."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33783-3_44"},{"key":"e_1_3_3_2_7_1","unstructured":"Yohann Cabon Naila Murray and Martin Humenberger. 2020. Virtual kitti 2. arXiv:https:\/\/arXiv.org\/abs\/2001.10773 (2020)."},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_23"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02126"},{"key":"e_1_3_3_2_10_1","unstructured":"Shouyuan Chen Sherman Wong Liangjian Chen and Yuandong Tian. 2023. Extending context window of large language models via positional interpolation. arXiv:https:\/\/arXiv.org\/abs\/2306.15595 (2023)."},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_3_2_12_1","volume-title":"NeurIPS","author":"Eigen David","year":"2014","unstructured":"David Eigen, Christian Puhrsch, and Rob Fergus. 2014. Depth map prediction from a single image using a multi-scale deep network. In NeurIPS."},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2019.00081"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00214"},{"key":"e_1_3_3_2_15_1","unstructured":"Xiao Fu Wei Yin Mu Hu Kaixuan Wang Yuexin Ma Ping Tan Shaojie Shen Dahua Lin and Xiaoxiao Long. 2024. GeoWizard: Unleashing the Diffusion Priors for 3D Geometry Estimation from a Single Image. arXiv:https:\/\/arXiv.org\/abs\/2403.12013 (2024)."},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"crossref","unstructured":"Andreas Geiger Philip Lenz Christoph Stiller and Raquel Urtasun. 2013. Vision meets Robotics: The KITTI Dataset. IJRR (2013).","DOI":"10.1177\/0278364913491297"},{"key":"e_1_3_3_2_17_1","unstructured":"Ming Gui Johannes\u00a0S Fischer Ulrich Prestel Pingchuan Ma Dmytro Kotovenko Olga Grebenkova Stefan\u00a0Andreas Baumann Vincent\u00a0Tao Hu and Bj\u00f6rn Ommer. 2024. DepthFM: Fast Monocular Depth Estimation with Flow Matching. arXiv:https:\/\/arXiv.org\/abs\/2403.13788 (2024)."},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"crossref","unstructured":"Mu Hu Wei Yin Chi Zhang Zhipeng Cai Xiaoxiao Long Hao Chen Kaixuan Wang Gang Yu Chunhua Shen and Shaojie Shen. 2024. Metric3d v2: A versatile monocular geometric foundation model for zero-shot metric depth and surface normal estimation. TPAMI (2024).","DOI":"10.1109\/TPAMI.2024.3444912"},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00193"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00298"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00907"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01645"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00981"},{"key":"e_1_3_3_2_25_1","unstructured":"LightwheelAI and LightwheelOcc contributors. 2024. LightwheelOcc: A 3D Occupancy Synthetic Dataset in Autonomous Driving. https:\/\/github.com\/OpenDriveLab\/LightwheelOcc."},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00482"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01556"},{"key":"e_1_3_3_2_28_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et\u00a0al. 2023. Dinov2: Learning robust visual features without supervision. TMLR (2023)."},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967590"},{"key":"e_1_3_3_2_30_1","unstructured":"Bowen Peng and Jeffrey Quesnelle. 2023. Ntk-aware scaled rope allows llama models to have extended (8k+) context size without any fine-tuning and minimal perplexity degradation."},{"key":"e_1_3_3_2_31_1","volume-title":"ICLR","author":"Peng Bowen","year":"2024","unstructured":"Bowen Peng, Jeffrey Quesnelle, Honglu Fan, and Enrico Shippole. 2024. YaRN: Efficient Context Window Extension of Large Language Models. In ICLR."},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.85"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"crossref","unstructured":"Luigi Piccinelli Christos Sakaridis Yung-Hsu Yang Mattia Segu Siyuan Li Wim Abbeloos and Luc\u00a0Van Gool. 2025. UniDepthV2: Universal Monocular Metric Depth Estimation Made Simpler. arXiv:https:\/\/arXiv.org\/abs\/2502.20110 (2025).","DOI":"10.1109\/CVPR52734.2025.00104"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00963"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"crossref","unstructured":"Ren\u00e9 Ranftl Katrin Lasinger David Hafner Konrad Schindler and Vladlen Koltun. 2022. Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer. TPAMI (2022).","DOI":"10.1109\/TPAMI.2020.3019967"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01073"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02127"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.377"},{"key":"e_1_3_3_2_40_1","unstructured":"Jianlin Su Murtadha Ahmed Yu Lu Shengfeng Pan Wen Bo and Yunfeng Liu. 2021. Roformer: Enhanced transformer with rotary position embedding. arXiv:https:\/\/arXiv.org\/abs\/2104.09864 (2021)."},{"key":"e_1_3_3_2_41_1","unstructured":"Yutao Sun Li Dong Barun Patra Shuming Ma Shaohan Huang Alon Benhaim Vishrav Chaudhary Xia Song and Furu Wei. 2022. A length-extrapolatable transformer. arXiv:https:\/\/arXiv.org\/abs\/2212.10554 (2022)."},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00883"},{"key":"e_1_3_3_2_43_1","volume-title":"NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention Is All You Need. In NeurIPS."},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00499"},{"key":"e_1_3_3_2_45_1","unstructured":"Kaixuan Wang and Shaojie Shen. 2019. Flow-Motion and Depth Network for Monocular Stereo and Beyond. arXiv:https:\/\/arXiv.org\/abs\/1909.05452 (2019)."},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME51207.2021.9428423"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00496"},{"key":"e_1_3_3_2_48_1","volume-title":"NeurIPS","author":"Wang Ruicheng","year":"2025","unstructured":"Ruicheng Wang, Sicheng Xu, Yue Dong, Yu Deng, Jianfeng Xiang, Zelong Lv, Guangzhong Sun, Xin Tong, and Jiaolong Yang. 2025c. MoGe-2: Accurate Monocular Geometry with Metric Scale and Sharp Details. In NeurIPS."},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01956"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9341801"},{"key":"e_1_3_3_2_51_1","unstructured":"Yiran Wang Min Shi Jiaqi Li Chaoyi Hong Zihao Huang Juewen Peng Zhiguo Cao Jianming Zhang Ke Xian and Guosheng Lin. 2024b. NVDS+: Towards Efficient and Versatile Neural Stabilizer for Video Depth Estimation. TPAMI (2024)."},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00868"},{"key":"e_1_3_3_2_53_1","volume-title":"ICLR","author":"Wang Yifan","year":"2026","unstructured":"Yifan Wang, Jianjun Zhou, Haoyi Zhu, Wenzheng Chang, Yang Zhou, Zizun Li, Junyi Chen, Jiangmiao Pang, Chunhua Shen, and Tong He. 2026. \u03c03: Scalable Permutation-Equivariant Visual Geometry Learning. In ICLR."},{"key":"e_1_3_3_2_54_1","volume-title":"ICLR","author":"Yang Honghui","year":"2025","unstructured":"Honghui Yang, Di Huang, Wei Yin, Chunhua Shen, Haifeng Liu, Xiaofei He, Binbin Lin, Wanli Ouyang, and Tong He. 2025. Depth Any Video with Scalable Synthetic Data. In ICLR."},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0688"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00008"},{"key":"e_1_3_3_2_58_1","unstructured":"Wei Yin Yifan Liu and Chunhua Shen. 2021a. Virtual Normal: Enforcing Geometric Constraints for Accurate and Robust Depth Prediction. TPAMI (2021)."},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00578"},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00830"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00027"},{"key":"e_1_3_3_2_62_1","volume-title":"ICLR","author":"Zhang Junyi","year":"2025","unstructured":"Junyi Zhang, Charles Herrmann, Junhwa Hur, Varun Jampani, Trevor Darrell, Forrester Cole, Deqing Sun, and Ming-Hsuan Yang. 2025. MonST3R: A Simple Approach for Estimating Geometry in the Presence of Motion. In ICLR."},{"key":"e_1_3_3_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01818"}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:13:20Z","timestamp":1784225600000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811149"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":62,"alternative-id":["10.1145\/3799902.3811149","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811149","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}