{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T16:29:46Z","timestamp":1781713786618,"version":"3.54.5"},"publisher-location":"Cham","reference-count":57,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733369","type":"print"},{"value":"9783031733376","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73337-6_4","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:02:27Z","timestamp":1730329347000},"page":"57-73","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Camera Height Doesn\u2019t Change: Unsupervised Training for\u00a0Metric Monocular Road-Scene Depth Estimation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9076-3152","authenticated-orcid":false,"given":"Genki","family":"Kinoshita","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3534-3447","authenticated-orcid":false,"given":"Ko","family":"Nishino","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"4_CR1","unstructured":"Bhat, S.F., Alhashim, I., Wonka, P.: AdaBins: depth estimation using adaptive bins. In: CVPR, pp. 4009\u20134018 (2021)"},{"key":"4_CR2","doi-asserted-by":"crossref","unstructured":"Casser, V., Pirk, S., Mahjourian, R., Angelova, A.: Unsupervised monocular depth and ego-motion learning with structure and semantics. In: CVPRW, pp. 381\u2013388 (2019)","DOI":"10.1109\/CVPRW.2019.00051"},{"key":"4_CR3","doi-asserted-by":"crossref","unstructured":"Chawla, H., Varma, A., Arani, E., Zonooz, B.: Multimodal scale consistency and awareness for monocular self-supervised depth estimation. In: ICRA, pp. 5140\u20135146 (2021)","DOI":"10.1109\/ICRA48506.2021.9561441"},{"key":"4_CR4","unstructured":"Chen, X., et al.: 3D object proposals for accurate object class detection. Adv. Neural Inform. Process. Syst. 28 (2015)"},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Chen, X., Li, T.H., Zhang, R., Li, G.: Frequency-aware self-supervised monocular depth estimation. In: WACV, pp. 5808\u20135817 (2023)","DOI":"10.1109\/WACV56688.2023.00576"},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"Cordts, M., et al.: The Cityscapes dataset for semantic urban scene understanding. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Ding, M., et al.: Learning depth-guided convolutions for monocular 3D object detection. In: CVPR, pp. 11669\u201311678 (2020)","DOI":"10.1109\/CVPR42600.2020.01169"},{"key":"4_CR8","doi-asserted-by":"crossref","unstructured":"Eigen, D., Fergus, R.: Predicting Depth, Surface Normals and Semantic Labels with a Common Multi-Scale Convolutional Architecture. In: ICCV. pp. 3213\u20133223 (2015)","DOI":"10.1109\/ICCV.2015.304"},{"key":"4_CR9","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Adv. Neural Inform. Process. Syst. 2, 2366\u20132374 (2014)"},{"issue":"3","key":"4_CR10","doi-asserted-by":"publisher","first-page":"736","DOI":"10.1109\/TRO.2018.2820722","volume":"34","author":"D Frost","year":"2018","unstructured":"Frost, D., Prisacariu, V., Murray, D.: Recovering stable scale in monocular SLAM using object-supplemented bundle adjustment. IEEE Trans. Rob. 34(3), 736\u2013747 (2018). https:\/\/doi.org\/10.1109\/TRO.2018.2820722","journal-title":"IEEE Trans. Rob."},{"key":"4_CR11","doi-asserted-by":"crossref","unstructured":"Garg, R., Bg, V.K., Carneiro, G., Reid, I.: Unsupervised CNN for single view depth estimation: geometry to the rescue. In: ECCV, pp. 740\u2013756 (2016)","DOI":"10.1007\/978-3-319-46484-8_45"},{"issue":"11","key":"4_CR12","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: the KITTI dataset. Inter. J. Robot. Res. (IJRR) 32(11), 1231\u20131237 (2013). https:\/\/doi.org\/10.1177\/0278364913491297","journal-title":"Inter. J. Robot. Res. (IJRR)"},{"key":"4_CR13","unstructured":"Geyer, J., et al.: A2D2: Audi Autonomous Driving Dataset. CoRR arXiv:abs\/2004.06320 (2020)"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: CVPR, pp. 6602\u20136611 (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"4_CR15","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Firman, M., Brostow, G.J.: Digging into self-supervised monocular depth estimation. In: ICCV, pp. 3827\u20133837 (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Ambrus, R., Pillai, S., Raventos, A., Gaidon, A.: 3D packing for self-supervised monocular depth estimation. In: CVPR, pp. 2482\u20132491 (2020)","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Hartley, R., Zisserman, A.: Multiple View Geometry in Computer Vision. Cambridge University Press, ISBN: 0521540518, second edn. (2004)","DOI":"10.1017\/CBO9780511811685"},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"He, M., Hui, L., Bian, Y., Ren, J., Xie, J., Yang, J.: RA-depth: resolution adaptive self-supervised monocular depth estimation. In: ECCV, pp. 565\u2013581 (2022)","DOI":"10.1007\/978-3-031-19812-0_33"},{"key":"4_CR19","doi-asserted-by":"crossref","unstructured":"Heylen, J., et al.: MonoCInIS: camera independent monocular 3D object detection using instance segmentation. In: ICCVW, pp. 923\u2013934 (2021)","DOI":"10.1109\/ICCVW54120.2021.00108"},{"key":"4_CR20","doi-asserted-by":"publisher","unstructured":"Hoiem, D., Efros, A.A., Hebert, M.: Putting objects in perspective. IJCV 80, 3\u201315 (2008). https:\/\/doi.org\/10.1007\/s11263-008-0137-5","DOI":"10.1007\/s11263-008-0137-5"},{"key":"4_CR21","unstructured":"Houston, J., et al.: One thousand and one hours: self-driving motion prediction dataset. In: Conference on Robot Learning, pp. 409\u2013418 (2021)"},{"key":"4_CR22","doi-asserted-by":"crossref","unstructured":"Jain, J., Li, J., Chiu, M.T., Hassani, A., Orlov, N., Shi, H.: OneFormer: one transformer to rule universal image segmentation. In: CVPR, pp. 2989\u20132998 (2023)","DOI":"10.1109\/CVPR52729.2023.00292"},{"key":"4_CR23","unstructured":"Kim, D., Ga, W., Ahn, P., Joo, D., Chun, S., Kim, J.: Global-Local Path Networks for Monocular Depth Estimation with Vertical CutDepth. CoRR arXiv:abs\/2201.07436 (2022)"},{"key":"4_CR24","unstructured":"Kingma, D.P., Ba, J.: Adam: A Method for Stochastic Optimization. CoRR arXiv:abs\/1412.6980 (2014)"},{"key":"4_CR25","doi-asserted-by":"crossref","unstructured":"Lyu, X., et al.: HR-Depth: high resolution self-supervised monocular depth estimation. In: AAAI. vol.\u00a035, pp. 2294\u20132301 (2021)","DOI":"10.1609\/aaai.v35i3.16329"},{"issue":"9","key":"4_CR26","doi-asserted-by":"publisher","first-page":"5408","DOI":"10.1109\/LRA.2023.3295254","volume":"8","author":"E Milli","year":"2023","unstructured":"Milli, E., Erkent, O., Y\u0131lmaz, A.E.: Multi-modal multi-task (3MT) road segmentation. IEEE Robot. Autom. Lett. 8(9), 5408\u20135415 (2023). https:\/\/doi.org\/10.1109\/LRA.2023.3295254","journal-title":"IEEE Robot. Autom. Lett."},{"key":"4_CR27","doi-asserted-by":"crossref","unstructured":"Ning, C., Gan, H.: Trap Attention: Monocular Depth Estimation with Manual Traps. In: CVPR. pp. 5033\u20135043 (2023)","DOI":"10.1109\/CVPR52729.2023.00487"},{"key":"4_CR28","unstructured":"Paszke, A., et al.: PyTorch: an imperative style, high-performance deep learning library. Adv. Neural Inform. Process. Syst, 8026\u20148037 (2019)"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Piccinelli, L., Sakaridis, C., Yu, F.: IDISC: internal discretization for monocular depth estimation. In: CVPR, pp. 21477\u201321487 (2023)","DOI":"10.1109\/CVPR52729.2023.02057"},{"key":"4_CR30","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: Imagenet: a large scale visual recognition challenge. IJCV 115, 211\u2013252 (2015)","journal-title":"IJCV"},{"key":"4_CR31","doi-asserted-by":"crossref","unstructured":"Sadat, A., Casas, S., Ren, M., Wu, X., Dhawan, P., Urtasun, R.: Perceive, predict, and plan: safe motion planning through interpretable semantic representations. In: ECCV, pp. 414\u2013430 (2020)","DOI":"10.1007\/978-3-030-58592-1_25"},{"key":"4_CR32","doi-asserted-by":"crossref","unstructured":"Schonberger, J.L., Frahm, J.M.: Structure-from-motion revisited. In: CVPR, pp. 4104\u20134113 (2016)","DOI":"10.1109\/CVPR.2016.445"},{"key":"4_CR33","doi-asserted-by":"crossref","unstructured":"Simonelli, A., Bul\u00f2, S.R., Porzi, L., Kontschieder, P., Ricci, E.: Are we missing confidence in pseudo-lidar methods for monocular 3D object detection? In: ICCV, pp. 3225\u20133233 (2021)","DOI":"10.1109\/ICCV48922.2021.00321"},{"key":"4_CR34","doi-asserted-by":"crossref","unstructured":"Spencer, J., Russell, C., Hadfield, S., Bowden, R.: Kick Back and Relax: Learning to Reconstruct the World by Watching SlowTV (2023)","DOI":"10.1109\/ICCV51070.2023.01445"},{"key":"4_CR35","doi-asserted-by":"crossref","unstructured":"Sucar, E., Hayet, J.B.: Bayesian scale estimation for monocular SLAM based on generic object detection for correcting scale drift. In: ICRA, pp. 5152\u20135158 (2018)","DOI":"10.1109\/ICRA.2018.8461178"},{"key":"4_CR36","doi-asserted-by":"crossref","unstructured":", Wagstaff, B., Kelly, J.: Self-supervised scale recovery for monocular depth and egomotion estimation. In: IROS, pp. 2620\u20132627 (2021)","DOI":"10.1109\/IROS51168.2021.9635938"},{"key":"4_CR37","doi-asserted-by":"crossref","unstructured":"Wang, C., Buenaposada, J.M., Zhu, R., Lucey, S.: Learning depth from monocular videos using direct methods. In: CVPR, pp. 2022\u20132030 (2018)","DOI":"10.1109\/CVPR.2018.00216"},{"key":"4_CR38","doi-asserted-by":"crossref","unstructured":"Wang, H., Cai, P., Fan, R., Sun, Y., Liu, M.: End-to-end interactive prediction and planning with optical flow distillation for autonomous driving. In: CVPRW, pp. 2229\u20132238 (2021)","DOI":"10.1109\/CVPRW53098.2021.00252"},{"key":"4_CR39","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chao, W.L., Garg, D., Hariharan, B., Campbell, M., Weinberger, K.Q.: Pseudo-LiDAR from visual depth estimation: bridging the gap in 3D object detection for autonomous driving. In: CVPR, pp. 8437\u20138445 (2019)","DOI":"10.1109\/CVPR.2019.00864"},{"key":"4_CR40","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Train in germany, test in the USA: making 3D object detectors generalize. In: CVPR, pp. 11710\u201311720 (2020)","DOI":"10.1109\/CVPR42600.2020.01173"},{"key":"4_CR41","doi-asserted-by":"crossref","unstructured":"Wang, Y., Yang, H., Cai, J., Wang, G., Wang, J., Huang, Y.: Unsupervised learning of depth and pose based on monocular camera and inertial measurement unit (IMU). In: ICRA, pp. 10010\u201310017 (2023)","DOI":"10.1109\/ICRA48891.2023.10160277"},{"issue":"4","key":"4_CR42","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A., Sheikh, H., Simoncelli, E.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13(4), 600\u2013612 (2004). https:\/\/doi.org\/10.1109\/TIP.2003.819861","journal-title":"IEEE Trans. Image Process."},{"key":"4_CR43","doi-asserted-by":"crossref","unstructured":"Watson, J., Mac\u00a0Aodha, O., Prisacariu, V., Brostow, G., Firman, M.: The temporal opportunist: self-supervised multi-frame monocular depth. In: CVPR, pp. 1164\u20131174 (2021)","DOI":"10.1109\/CVPR46437.2021.00122"},{"key":"4_CR44","unstructured":"Wilson, B., et al.: Argoverse 2: next generation datasets for self-driving perception and forecasting. In: NeurIPS Datasets and Benchmarks (2021)"},{"issue":"4","key":"4_CR45","doi-asserted-by":"publisher","first-page":"11998","DOI":"10.1109\/LRA.2022.3210298","volume":"7","author":"J Xiang","year":"2022","unstructured":"Xiang, J., Wang, Y., An, L., Liu, H., Wang, Z., Liu, J.: Visual attention-based self-supervised absolute depth estimation using geometric priors in autonomous driving. IEEE Robot. Auto. Lett. 7(4), 11998\u201312005 (2022). https:\/\/doi.org\/10.1109\/LRA.2022.3210298","journal-title":"IEEE Robot. Auto. Lett."},{"key":"4_CR46","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, P., Wang, Y., Xu, W., Nevatia, R.: LEGO: learning edge with geometry all at once by watching videos. In: CVPR, pp. 225\u2013234 (2018)","DOI":"10.1109\/CVPR.2018.00031"},{"key":"4_CR47","doi-asserted-by":"crossref","unstructured":"Yin, W., Zhang, C., Chen, H., Cai, Z., Yu, G., Wang, K., Chen, X., Shen, C.: Metric3D: Towards Zero-Shot Metric 3D Prediction from a Single Image. In: ICCV. pp. 9043\u20139053 (2023)","DOI":"10.1109\/ICCV51070.2023.00830"},{"key":"4_CR48","doi-asserted-by":"crossref","unstructured":"Yin, Z., Shi, J.: GeoNet: unsupervised learning of dense depth, optical flow and camera pose. In: CVPR, pp. 1983\u20131992 (2018)","DOI":"10.1109\/CVPR.2018.00212"},{"key":"4_CR49","unstructured":"You, Y., et al.: Pseudo-LiDAR++: accurate depth for 3D object detection in autonomous driving. In: ICLR (2020)"},{"key":"4_CR50","doi-asserted-by":"crossref","unstructured":"Yuan, W., Gu, X., Dai, Z., Zhu, S., Tan, P.: Neural window fully-connected CRFs for monocular depth estimation. In: CVPR, pp. 3916\u20133925 (2022)","DOI":"10.1109\/CVPR52688.2022.00389"},{"key":"4_CR51","doi-asserted-by":"crossref","unstructured":"Zeng, W., et al.: End-to-end interpretable neural motion planner. In: CVPR, pp. 8652\u20138661 (2019)","DOI":"10.1109\/CVPR.2019.00886"},{"key":"4_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, N., Nex, F., Vosselman, G., Kerle, N.: Lite-Mono: A Lightweight CNN and Transformer Architecture for Self-Supervised Monocular Depth Estimation. In: CVPR. pp. 18537\u201318546 (2023)","DOI":"10.1109\/CVPR52729.2023.01778"},{"key":"4_CR53","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhang, J., Tao, D.: Towards scale-aware, robust, and generalizable unsupervised monocular depth estimation by integrating IMU motion dynamics. In: ECCV, pp. 143\u2013160 (2022)","DOI":"10.1007\/978-3-031-19839-7_9"},{"key":"4_CR54","doi-asserted-by":"crossref","unstructured":"Zhang, S., Li, X., Liu, Y., Fu, H.: Scale-aware insertion of virtual objects in monocular videos. In: ISMAR, pp. 36\u201344 (2020)","DOI":"10.1109\/ISMAR50242.2020.00022"},{"key":"4_CR55","unstructured":"Zhou, H., Greenwood, D., Taylor, S.: Self-supervised monocular depth estimation with internal feature fusion. In: BMVC (2021)"},{"key":"4_CR56","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: CVPR, pp. 6612\u20136619 (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"4_CR57","doi-asserted-by":"crossref","unstructured":"Zhu, R., et al.: Single view metrology in the wild. In: ECCV, pp. 316\u2013333 (2020)","DOI":"10.1007\/978-3-030-58621-8_19"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73337-6_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:03:13Z","timestamp":1730329393000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73337-6_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733369","9783031733376"],"references-count":57,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73337-6_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}