{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T21:55:53Z","timestamp":1777758953074,"version":"3.51.4"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s00371-026-04445-7","type":"journal-article","created":{"date-parts":[[2026,4,5]],"date-time":"2026-04-05T15:44:46Z","timestamp":1775403886000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhanced monocular depth estimation via semantic fusion and planar constraints"],"prefix":"10.1007","volume":"42","author":[{"given":"Wenhao","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunyu","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhensong","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shoubiao","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiao","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,5]]},"reference":[{"key":"4445_CR1","doi-asserted-by":"crossref","unstructured":"Yan, J., Zhao, H., Bu, P., Jin, Y.: Channel-wise attention-based network for self-supervised monocular depth estimation. In: 2021 International Conference on 3D Vision (3DV), 464\u2013473 (2021). IEEE","DOI":"10.1109\/3DV53792.2021.00056"},{"key":"4445_CR2","doi-asserted-by":"crossref","unstructured":"Lyu, X., Liu, L., Wang, M., Kong, X., Liu, L., Liu, Y., Chen, X., Yuan, Y.: Hr-depth: High resolution self-supervised monocular depth estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, 35, 2294\u20132301 (2021)","DOI":"10.1609\/aaai.v35i3.16329"},{"key":"4445_CR3","doi-asserted-by":"crossref","unstructured":"Bae, J., Moon, S., Im, S.: Deep digging into the generalization of self-supervised monocular depth estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, 37, 187\u2013196 (2023)","DOI":"10.1609\/aaai.v37i1.25090"},{"issue":"4","key":"4445_CR4","doi-asserted-by":"publisher","first-page":"10969","DOI":"10.1109\/LRA.2022.3196781","volume":"7","author":"D Han","year":"2022","unstructured":"Han, D., Shin, J., Kim, N., Hwang, S., Choi, Y.: Transdssl: Transformer based depth estimation via self-supervised learning. IEEE Robotics and Automation Letters 7(4), 10969\u201310976 (2022)","journal-title":"IEEE Robotics and Automation Letters"},{"key":"4445_CR5","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Zhang, Y., Yu, Y., Song, Z., Tang, C.: Tinydepth: Lightweight self-supervised monocular depth estimation based on transformer. Eng. Appl. Artif. Intell. 138, 109313 (2024)","DOI":"10.1016\/j.engappai.2024.109313"},{"key":"4445_CR6","doi-asserted-by":"crossref","unstructured":"Ma, J., Lei, X., Liu, N., Zhao, X., Pu, S.: Towards comprehensive representation enhancement in semantics-guided self-supervised monocular depth estimation. In: European Conference on Computer Vision, 304\u2013321 (2022). Springer","DOI":"10.1007\/978-3-031-19769-7_18"},{"key":"4445_CR7","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Advances in neural information processing systems 27 (2014)"},{"key":"4445_CR8","unstructured":"Bhat, S.F., Alhashim, I., Wonka, P.: Adabins: Depth estimation using adaptive bins. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4009\u20134018 (2021)"},{"key":"4445_CR9","doi-asserted-by":"crossref","unstructured":"Zhao, S., Fu, H., Gong, M., Tao, D.: Geometry-aware symmetric domain adaptation for monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 9788\u20139798 (2019)","DOI":"10.1109\/CVPR.2019.01002"},{"key":"4445_CR10","unstructured":"Kim, D., Ka, W., Ahn, P., Joo, D., Chun, S., Kim, J.: Global-local path networks for monocular depth estimation with vertical cutdepth. arXiv preprint arXiv:2201.07436 (2022)"},{"issue":"5","key":"4445_CR11","doi-asserted-by":"publisher","first-page":"418","DOI":"10.1016\/j.vrih.2022.08.005","volume":"4","author":"C Li","year":"2022","unstructured":"Li, C., Yi, R., Ali, S.G., Ma, L., Wu, E., Wang, J., Mao, L., Sheng, B.: Radepthnet: reflectance-aware monocular depth estimation. Virtual Reality & Intelligent Hardware 4(5), 418\u2013431 (2022)","journal-title":"Virtual Reality & Intelligent Hardware"},{"key":"4445_CR12","doi-asserted-by":"crossref","unstructured":"Wang, J., Lu, X., Bennamoun, M., Sheng, B.: Non-rigid point cloud registration via anisotropic hybrid field harmonization. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)","DOI":"10.1109\/TPAMI.2025.3572584"},{"key":"4445_CR13","doi-asserted-by":"crossref","unstructured":"Garg, R., Bg, V.K., Carneiro, G., Reid, I.: Unsupervised cnn for single view depth estimation: Geometry to the rescue. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part VIII 14, 740\u2013756 (2016). Springer","DOI":"10.1007\/978-3-319-46484-8_45"},{"key":"4445_CR14","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 270\u2013279 (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"4445_CR15","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 1851\u20131858 (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"4445_CR16","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Firman, M., Brostow, G.J.: Digging into self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 3828\u20133838 (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"4445_CR17","doi-asserted-by":"crossref","unstructured":"Shu, C., Yu, K., Duan, Z., Yang, K.: Feature-metric loss for self-supervised learning of depth and egomotion. In: European Conference on Computer Vision, 572\u2013588 (2020). Springer","DOI":"10.1007\/978-3-030-58529-7_34"},{"key":"4445_CR18","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Ambrus, R., Pillai, S., Raventos, A., Gaidon, A.: 3d packing for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2485\u20132494 (2020)","DOI":"10.1109\/CVPR42600.2020.00256"},{"issue":"1","key":"4445_CR19","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1007\/s00371-024-03319-0","volume":"41","author":"X Qin","year":"2025","unstructured":"Qin, X., Li, X., Li, M., Zheng, H., Xu, X.: Self-supervised single-image 3d face reconstruction method based on attention mechanism and attribute refinement. Vis. Comput. 41(1), 209\u2013227 (2025)","journal-title":"Vis. Comput."},{"key":"4445_CR20","doi-asserted-by":"crossref","unstructured":"Watson, J., Mac\u00a0Aodha, O., Prisacariu, V., Brostow, G., Firman, M.: The temporal opportunist: Self-supervised multi-frame monocular depth. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 1164\u20131174 (2021)","DOI":"10.1109\/CVPR46437.2021.00122"},{"key":"4445_CR21","doi-asserted-by":"crossref","unstructured":"Liu, J., Kong, L., Li, B., Wang, Z., Gu, H., Chen, J.: Mono-vifi: A unified learning framework for self-supervised single and multi-frame monocular depth estimation. In: European Conference on Computer Vision, 90\u2013107 (2024). Springer","DOI":"10.1007\/978-3-031-72995-9_6"},{"key":"4445_CR22","doi-asserted-by":"crossref","unstructured":"He, M., Hui, L., Bian, Y., Ren, J., Xie, J., Yang, J.: Ra-depth: Resolution adaptive self-supervised monocular depth estimation. In: European Conference on Computer Vision, 565\u2013581 (2022). Springer","DOI":"10.1007\/978-3-031-19812-0_33"},{"key":"4445_CR23","doi-asserted-by":"crossref","unstructured":"Zhang, N., Nex, F., Vosselman, G., Kerle, N.: Lite-mono: A lightweight cnn and transformer architecture for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 18537\u201318546 (2023)","DOI":"10.1109\/CVPR52729.2023.01778"},{"key":"4445_CR24","doi-asserted-by":"crossref","unstructured":"Wang, R., Yu, Z., Gao, S.: Planedepth: Self-supervised depth estimation via orthogonal planes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 21425\u201321434 (2023)","DOI":"10.1109\/CVPR52729.2023.02052"},{"key":"4445_CR25","doi-asserted-by":"crossref","unstructured":"Moon, J., Bello, J.L.G., Kwon, B., Kim, M.: From-ground-to-objects: Coarse-to-fine self-supervised monocular depth estimation of dynamic objects with ground contact prior. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 10519\u201310529 (2024)","DOI":"10.1109\/CVPR52733.2024.01001"},{"key":"4445_CR26","doi-asserted-by":"crossref","unstructured":"Chen, X., Zhang, R., Jiang, J., Wang, Y., Li, G., Li, T.H.: Self-supervised monocular depth estimation: Solving the edge-fattening problem. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 5776\u20135786 (2023)","DOI":"10.1109\/WACV56688.2023.00573"},{"issue":"15","key":"4445_CR27","doi-asserted-by":"publisher","first-page":"18167","DOI":"10.1007\/s10489-022-03401-x","volume":"52","author":"T Gao","year":"2022","unstructured":"Gao, T., Wei, W., Cai, Z., Fan, Z., Xie, S.Q., Wang, X., Yu, Q.: Ci-net: A joint depth estimation and semantic segmentation network using contextual information. Appl. Intell. 52(15), 18167\u201318186 (2022)","journal-title":"Appl. Intell."},{"key":"4445_CR28","doi-asserted-by":"publisher","first-page":"251","DOI":"10.1016\/j.neucom.2021.01.126","volume":"440","author":"L He","year":"2021","unstructured":"He, L., Lu, J., Wang, G., Song, S., Zhou, J.: Sosd-net: Joint semantic object segmentation and depth estimation from monocular images. Neurocomputing 440, 251\u2013263 (2021)","journal-title":"Neurocomputing"},{"key":"4445_CR29","doi-asserted-by":"crossref","unstructured":"Zhou, H., Greenwood, D., Taylor, S.: Self-supervised monocular depth estimation with internal feature fusion. arXiv preprint arXiv:2110.09482 (2021)","DOI":"10.5244\/C.35.208"},{"key":"4445_CR30","doi-asserted-by":"crossref","unstructured":"Klingner, M., Term\u00f6hlen, J.-A., Mikolajczyk, J., Fingscheidt, T.: Self-supervised monocular depth estimation: Solving the dynamic object problem by semantic guidance. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XX 16, 582\u2013600 (2020). Springer","DOI":"10.1007\/978-3-030-58565-5_35"},{"key":"4445_CR31","doi-asserted-by":"crossref","unstructured":"Jung, H., Park, E., Yoo, S.: Fine-grained semantics-aware representation enhancement for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 12642\u201312652 (2021)","DOI":"10.1109\/ICCV48922.2021.01241"},{"issue":"1","key":"4445_CR32","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1007\/s00371-024-03320-7","volume":"41","author":"X Lei","year":"2025","unstructured":"Lei, X., Chen, Z., Yu, Z., Jiang, Z.: Benet: boundary-enhanced network for real-time semantic segmentation. Vis. Comput. 41(1), 229\u2013241 (2025)","journal-title":"Vis. Comput."},{"key":"4445_CR33","doi-asserted-by":"crossref","unstructured":"Lee, Y., Kim, J., Willette, J., Hwang, S.J.: Mpvit: Multi-path vision transformer for dense prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 7287\u20137296 (2022)","DOI":"10.1109\/CVPR52688.2022.00714"},{"key":"4445_CR34","doi-asserted-by":"crossref","unstructured":"Zhao, C., Zhang, Y., Poggi, M., Tosi, F., Guo, X., Zhu, Z., Huang, G., Tang, Y., Mattoccia, S.: Monovit: Self-supervised monocular depth estimation with a vision transformer. In: 2022 International Conference on 3D Vision (3DV), 668\u2013678 (2022). IEEE","DOI":"10.1109\/3DV57658.2022.00077"},{"key":"4445_CR35","doi-asserted-by":"crossref","unstructured":"Jiang, H., Fang, Z., Shao, X., Jiang, X., Hwang, J.-N.: Lidut-depth: A lightweight self-supervised depth estimation model featuring dynamic upsampling and triplet loss optimization. In: International Conference on Pattern Recognition, 176\u2013189 (2025). Springer","DOI":"10.1007\/978-3-031-78444-6_12"},{"key":"4445_CR36","doi-asserted-by":"crossref","unstructured":"Shim, D., Kim, H.J.: Swindepth: Unsupervised depth estimation using monocular sequences via swin transformer and densely cascaded network. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), 4983\u20134990 (2023). IEEE","DOI":"10.1109\/ICRA48891.2023.10160657"},{"issue":"6","key":"4445_CR37","doi-asserted-by":"publisher","first-page":"4781","DOI":"10.1007\/s10489-024-05414-0","volume":"54","author":"Y Zhou","year":"2024","unstructured":"Zhou, Y., Zhang, C., Deng, L., Fu, J., Li, H., Xu, Z., Zhang, J.: Resolution-sensitive self-supervised monocular absolute depth estimation. Appl. Intell. 54(6), 4781\u20134793 (2024)","journal-title":"Appl. Intell."},{"key":"4445_CR38","unstructured":"Bello, J.L.G., Moon, J., Kim, M.: Self-supervised monocular depth estimation with positional shift depth variance and adaptive disparity quantization. IEEE Transactions on Image Processing (2024)"},{"issue":"11","key":"4445_CR39","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: The kitti dataset. The international journal of robotics research 32(11), 1231\u20131237 (2013)","journal-title":"The international journal of robotics research"},{"key":"4445_CR40","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Sapra, K., Reda, F.A., Shih, K.J., Newsam, S., Tao, A., Catanzaro, B.: Improving semantic segmentation via video propagation and label relaxation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8856\u20138865 (2019)","DOI":"10.1109\/CVPR.2019.00906"},{"issue":"5","key":"4445_CR41","doi-asserted-by":"publisher","first-page":"824","DOI":"10.1109\/TPAMI.2008.132","volume":"31","author":"A Saxena","year":"2008","unstructured":"Saxena, A., Sun, M., Ng, A.Y.: Make3d: Learning 3d scene structure from a single still image. IEEE Trans. Pattern Anal. Mach. Intell. 31(5), 824\u2013840 (2008)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4445_CR42","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from rgbd images. In: Computer Vision\u2013ECCV 2012: 12th European Conference on Computer Vision, Florence, Italy, October 7-13, 2012, Proceedings, Part V 12, 746\u2013760 (2012). Springer","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"4445_CR43","doi-asserted-by":"crossref","unstructured":"Uhrig, J., Schneider, N., Schneider, L., Franke, U., Brox, T., Geiger, A.: lsparsity invariant cnns. In: 2017 International Conference on 3D Vision (3DV), 11\u201320 (2017). IEEE","DOI":"10.1109\/3DV.2017.00012"},{"key":"4445_CR44","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., Fei-Fei, L.: Imagenet: A large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, 248\u2013255 (2009). Ieee","DOI":"10.1109\/CVPR.2009.5206848"},{"issue":"12","key":"4445_CR45","first-page":"1","volume":"21","author":"Z Song","year":"2025","unstructured":"Song, Z., Zhu, R., Wang, J., Wang, C., He, J., Deng, J., Yang, W., Zhang, T.: Er-depth: Enhancing the robustness of self-supervised monocular depth estimation in challenging scenes. ACM Trans. Multimed. Comput. Commun. Appl. 21(12), 1\u201323 (2025)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04445-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04445-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04445-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T13:17:23Z","timestamp":1777468643000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04445-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":45,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["4445"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04445-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]},"assertion":[{"value":"14 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"241"}}