{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:58:16Z","timestamp":1776931096972,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61936015"],"award-info":[{"award-number":["61936015"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai","doi-asserted-by":"publisher","award":["24ZR1430600"],"award-info":[{"award-number":["24ZR1430600"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763991","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:30:41Z","timestamp":1765211441000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["MODepth: Benchmarking Mobile Multi-frame Monocular Depth Estimation with Optical Image Stabilization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9024-3692","authenticated-orcid":false,"given":"Yu","family":"Lu","sequence":"first","affiliation":[{"name":"School of Computer Science, Shanghai Jiao Tong University, Shanghai, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2531-0107","authenticated-orcid":false,"given":"Hao","family":"Pan","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Shanghai, China and School of Computer Science, Shanghai Jiao Tong University, Shanghai, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2190-0919","authenticated-orcid":false,"given":"Dian","family":"Ding","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7552-9468","authenticated-orcid":false,"given":"Jiatong","family":"Ding","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8481-2644","authenticated-orcid":false,"given":"Yongjian","family":"Fu","sequence":"additional","affiliation":[{"name":"Central South University, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0782-4953","authenticated-orcid":false,"given":"Yi-Chao","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2782-183X","authenticated-orcid":false,"given":"Ju","family":"Ren","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1617-3593","authenticated-orcid":false,"given":"Guangtao","family":"Xue","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"Dosovitskiy Alexey. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/ 2010.11929 (2020)."},{"key":"e_1_3_3_2_3_1","first-page":"4009","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","author":"Bhat Shariq\u00a0Farooq","year":"2021","unstructured":"Shariq\u00a0Farooq Bhat, Ibraheem Alhashim, and Peter Wonka. 2021. Adabins: Depth estimation using adaptive bins. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 4009\u20134018."},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"crossref","unstructured":"Jia-Wang Bian Huangying Zhan Naiyan Wang Tat-Jun Chin Chunhua Shen and Ian Reid. 2021a. Auto-rectify network for unsupervised indoor depth estimation. IEEE transactions on pattern analysis and machine intelligence 44 12 (2021) 9802\u20139813.","DOI":"10.1109\/TPAMI.2021.3136220"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Jia-Wang Bian Huangying Zhan Naiyan Wang Zhichao Li Le Zhang Chunhua Shen Ming-Ming Cheng and Ian Reid. 2021b. Unsupervised scale-consistent depth learning from video. International Journal of Computer Vision 129 9 (2021) 2548\u20132564.","DOI":"10.1007\/s11263-021-01484-6"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"crossref","unstructured":"Brent Cardani. 2006. Optical image stabilization for digital cameras. IEEE Control Systems Magazine 26 2 (2006) 21\u201322.","DOI":"10.1109\/MCS.2006.1615267"},{"key":"e_1_3_3_2_7_1","unstructured":"Xuelian Cheng Yiran Zhong Mehrtash Harandi Yuchao Dai Xiaojun Chang Hongdong Li Tom Drummond and Zongyuan Ge. 2020. Hierarchical neural architecture search for deep stereo matching. Advances in neural information processing systems 33 (2020) 22158\u201322169."},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00287"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_3_2_11_1","first-page":"1","volume-title":"Proceedings of the 1st Annual Conference on Robot Learning","author":"Dosovitskiy Alexey","year":"2017","unstructured":"Alexey Dosovitskiy, German Ros, Felipe Codevilla, Antonio Lopez, and Vladlen Koltun. 2017. CARLA: An Open Urban Driving Simulator. In Proceedings of the 1st Annual Conference on Robot Learning. 1\u201316."},{"key":"e_1_3_3_2_12_1","unstructured":"Chao Fan Zhenyu Yin Yue Li and Feiqing Zhang. 2023. Deeper into Self-Supervised Monocular Indoor Depth Estimation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.01283 (2023)."},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093334"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00214"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372224.3419210"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00393"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_30"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2019.00116"},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298762"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01255"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00083"},{"key":"e_1_3_3_2_23_1","unstructured":"Neel Joshi and C\u00a0Lawrence Zitnick. 2014. Micro-baseline stereo. Microsoft Research Technical Report (2014)."},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.238"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2016.32"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01243"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.365"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00032"},{"key":"e_1_3_3_2_29_1","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.05101 (2017)."},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3666025.3699371"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560523"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547964"},{"key":"e_1_3_3_2_33_1","unstructured":"Adam Paszke Sam Gross Francisco Massa Adam Lerer James Bradbury Gregory Chanan Trevor Killeen Zeming Lin Natalia Gimelshein Luca Antiga et\u00a0al. 2019. Pytorch: An imperative style high-performance deep learning library. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02057"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00963"},{"key":"e_1_3_3_2_36_1","unstructured":"Santhosh\u00a0K Ramakrishnan Aaron Gokaslan Erik Wijmans Oleksandr Maksymets Alex Clegg John Turner Eric Undersander Wojciech Galuba Andrew Westbury Angel\u00a0X Chang et\u00a0al. 2021. Habitat-matterport 3d dataset (hm3d): 1000 large-scale 3d environments for embodied ai. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2109.08238 (2021)."},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_3_2_39_1","first-page":"4049","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Saragadam Vishwanath","year":"2019","unstructured":"Vishwanath Saragadam, Jian Wang, Mohit Gupta, and Shree Nayar. 2019. Micro-baseline structured light. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 4049\u20134058."},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"},{"key":"e_1_3_3_2_41_1","unstructured":"Julian Straub Thomas Whelan Lingni Ma Yufan Chen Erik Wijmans Simon Green Jakob\u00a0J Engel Raul Mur-Artal Carl Ren Shobhit Verma et\u00a0al. 2019. The Replica dataset: A digital replica of indoor spaces. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1906.05797 (2019)."},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"crossref","unstructured":"J Su Y Lu S Pan A Murtadha B Wen and Y\u00a0Liu Roformer. 2023. Enhanced transformer with rotary position embedding. 2021. DOI: https:\/\/doi. org\/10.1016\/j. neucom (2023).","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_3_2_43_1","unstructured":"Andrew Szot Alexander Clegg Eric Undersander Erik Wijmans Yili Zhao John Turner Noah Maestre Mustafa Mukadam Devendra\u00a0Singh Chaplot Oleksandr Maksymets et\u00a0al. 2021. Habitat 2.0: Training home assistants to rearrange their habitat. Advances in neural information processing systems 34 (2021) 251\u2013266."},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/EuroSP.2017.42"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"crossref","unstructured":"Zhou Wang Alan\u00a0C Bovik Hamid\u00a0R Sheikh and Eero\u00a0P Simoncelli. 2004. Image quality assessment: from error visibility to structural similarity. IEEE transactions on image processing 13 4 (2004) 600\u2013612.","DOI":"10.1109\/TIP.2003.819861"},{"key":"e_1_3_3_2_47_1","unstructured":"Philippe Weinzaepfel Vincent Leroy Thomas Lucas Romain Br\u00e9gier Yohann Cabon Vaibhav Arora Leonid Antsfeld Boris Chidlovskii Gabriela Csurka and J\u00e9r\u00f4me Revaud. 2022. Croco: Self-supervised pre-training for 3d vision tasks by cross-view completion. Advances in Neural Information Processing Systems 35 (2022) 3502\u20133516."},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01647"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00379"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"crossref","unstructured":"Yao Yao Zixin Luo Shiwei Li Tian Fang and Long Quan. 2018. MVSNet: Depth Inference for Unstructured Multi-view Stereo. European Conference on Computer Vision (ECCV) (2018).","DOI":"10.1007\/978-3-030-01237-3_47"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"crossref","unstructured":"Yao Yao Zixin Luo Shiwei Li Tianwei Shen Tian Fang and Long Quan. 2019. Recurrent MVSNet for High-resolution Multi-view Stereo Depth Inference. Computer Vision and Pattern Recognition (CVPR) (2019).","DOI":"10.1109\/CVPR.2019.00567"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00578"},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58586-0_13"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-96530-3"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00917"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00871"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763991","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:32:18Z","timestamp":1765251138000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763991"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":56,"alternative-id":["10.1145\/3757377.3763991","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763991","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}