{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T17:56:14Z","timestamp":1784397374906,"version":"3.55.0"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2024,6,22]],"date-time":"2024-06-22T00:00:00Z","timestamp":1719014400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,22]],"date-time":"2024-06-22T00:00:00Z","timestamp":1719014400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2024,7]]},"DOI":"10.1007\/s00371-024-03498-w","type":"journal-article","created":{"date-parts":[[2024,6,22]],"date-time":"2024-06-22T20:10:26Z","timestamp":1719087026000},"page":"4955-4967","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Multi-feature fusion enhanced monocular depth estimation with boundary awareness"],"prefix":"10.1007","volume":"40","author":[{"given":"Chao","family":"Song","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingjie","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Frederick W. B.","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoyi","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuliang","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bailin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,6,22]]},"reference":[{"key":"3498_CR1","doi-asserted-by":"crossref","unstructured":"Bae, J., Moon, S., Im, S.: Deep digging into the generalization of self-supervised monocular depth estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 187\u2013196 (2023)","DOI":"10.1609\/aaai.v37i1.25090"},{"key":"3498_CR2","doi-asserted-by":"crossref","unstructured":"Chen, P.Y., Liu, A.H., Liu, Y.C., Wang, Y.C.F.: Towards scene understanding: Unsupervised monocular depth estimation with semantic-aware representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2624\u20132632 (2019)","DOI":"10.1109\/CVPR.2019.00273"},{"key":"3498_CR3","doi-asserted-by":"crossref","unstructured":"Chen, X., Zhang, R., Jiang, J., Wang, Y., Li, G., Li, T.H.: Self-supervised monocular depth estimation: Solving the edge-fattening problem. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 5776\u20135786 (2023)","DOI":"10.1109\/WACV56688.2023.00573"},{"key":"3498_CR4","unstructured":"Choi, J., Jung, D., Lee, D., Kim, C.: Safenet: Self-supervised monocular depth estimation with semantic-aware feature extraction. In: Thirty-fourth Conference on Neural Information Processing Systems, NIPS 2020. NeurIPS (2020)"},{"key":"3498_CR5","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations (2020)"},{"key":"3498_CR6","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. Adv. Neural Inform. Process. Syst. 27 (2014)"},{"issue":"11","key":"3498_CR7","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., Urtasun, R.: Vision meets robotics: the kitti dataset. Int. J. Robot. Res. 32(11), 1231\u20131237 (2013)","journal-title":"Int. J. Robot. Res."},{"key":"3498_CR8","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 270\u2013279 (2017)","DOI":"10.1109\/CVPR.2017.699"},{"key":"3498_CR9","doi-asserted-by":"crossref","unstructured":"Godard, C., Mac\u00a0Aodha, O., Firman, M., Brostow, G.J.: Digging into self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3828\u20133838 (2019)","DOI":"10.1109\/ICCV.2019.00393"},{"key":"3498_CR10","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Ambrus, R., Pillai, S., Raventos, A., Gaidon, A.: 3d packing for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2485\u20132494 (2020)","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"3498_CR11","doi-asserted-by":"crossref","unstructured":"Hirschmuller, H.: Accurate and efficient stereo processing by semi-global matching and mutual information. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR\u201905), vol.\u00a02, pp. 807\u2013814. IEEE (2005)","DOI":"10.1109\/CVPR.2005.56"},{"key":"3498_CR12","doi-asserted-by":"crossref","unstructured":"Johnston, A., Carneiro, G.: Self-supervised monocular trained depth estimation using self-attention and discrete disparity volume. In: Proceedings of the IEEE\/cvf Conference on Computer Vision and Pattern Recognition, pp. 4756\u20134765 (2020)","DOI":"10.1109\/CVPR42600.2020.00481"},{"key":"3498_CR13","doi-asserted-by":"crossref","unstructured":"Jung, H., Park, E., Yoo, S.: Fine-grained semantics-aware representation enhancement for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12,642\u201312,652 (2021)","DOI":"10.1109\/ICCV48922.2021.01241"},{"key":"3498_CR14","doi-asserted-by":"crossref","unstructured":"Kendall, A., Gal, Y., Cipolla, R.: Multi-task learning using uncertainty to weigh losses for scene geometry and semantics. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7482\u20137491 (2018)","DOI":"10.1109\/CVPR.2018.00781"},{"key":"3498_CR15","doi-asserted-by":"crossref","unstructured":"Klingner, M., Term\u00f6hlen, J.A., Mikolajczyk, J., Fingscheidt, T.: Self-supervised monocular depth estimation: solving the dynamic object problem by semantic guidance. In: European Conference on Computer Vision, pp. 582\u2013600. Springer (2020)","DOI":"10.1007\/978-3-030-58565-5_35"},{"key":"3498_CR16","doi-asserted-by":"crossref","unstructured":"Laina, I., Rupprecht, C., Belagiannis, V., Tombari, F., Navab, N.: Deeper depth prediction with fully convolutional residual networks. In: 2016 Fourth International Conference on 3D Vision (3DV), pp. 239\u2013248. IEEE (2016)","DOI":"10.1109\/3DV.2016.32"},{"key":"3498_CR17","doi-asserted-by":"crossref","unstructured":"Lee, Y., Kim, J., Willette, J., Hwang, S.J.: Mpvit: Multi-path vision transformer for dense prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7287\u20137296 (2022)","DOI":"10.1109\/CVPR52688.2022.00714"},{"key":"3498_CR18","doi-asserted-by":"crossref","unstructured":"Lyu, X., Liu, L., Wang, M., Kong, X., Liu, L., Liu, Y., Chen, X., Yuan, Y.: Hr-depth: high resolution self-supervised monocular depth estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 2294\u20132301 (2021)","DOI":"10.1609\/aaai.v35i3.16329"},{"key":"3498_CR19","doi-asserted-by":"crossref","unstructured":"Peng, R., Wang, R., Lai, Y., Tang, L., Cai, Y.: Excavating the potential capacity of self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15,560\u201315,569 (2021)","DOI":"10.1109\/ICCV48922.2021.01527"},{"key":"3498_CR20","doi-asserted-by":"crossref","unstructured":"Shu, C., Yu, K., Duan, Z., Yang, K.: Feature-metric loss for self-supervised learning of depth and egomotion. In: European Conference on Computer Vision, pp. 572\u2013588. Springer (2020)","DOI":"10.1007\/978-3-030-58529-7_34"},{"issue":"11","key":"3498_CR21","doi-asserted-by":"publisher","first-page":"4381","DOI":"10.1109\/TCSVT.2021.3049869","volume":"31","author":"M Song","year":"2021","unstructured":"Song, M., Lim, S., Kim, W.: Monocular depth estimation using laplacian pyramid-based depth residuals. IEEE Transact. Circ. Syst. Video Technol. 31(11), 4381\u20134393 (2021)","journal-title":"IEEE Transact. Circ. Syst. Video Technol."},{"key":"3498_CR22","doi-asserted-by":"crossref","unstructured":"Sun, L., Bian, J.W., Zhan, H., Yin, W., Reid, I., Shen, C.: Sc-depthv3: Robust self-supervised monocular depth estimation for dynamic scenes. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI) (2023)","DOI":"10.1109\/TPAMI.2023.3322549"},{"key":"3498_CR23","doi-asserted-by":"crossref","unstructured":"Sun, Q., Tang, Y., Zhang, C., Zhao, C., Qian, F., Kurths, J.: Unsupervised estimation of monocular depth and vo in dynamic environments via hybrid masks. IEEE Transact. Neural Netw. Learn. Syst. 33(5), 2023\u20132033 (2022)","DOI":"10.1109\/TNNLS.2021.3100895"},{"key":"3498_CR24","doi-asserted-by":"crossref","unstructured":"Tosi, F., Aleotti, F., Poggi, M., Mattoccia, S.: Learning monocular depth estimation infusing traditional stereo knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9799\u20139809 (2019)","DOI":"10.1109\/CVPR.2019.01003"},{"key":"3498_CR25","doi-asserted-by":"crossref","unstructured":"Wang, C., Buenaposada, J.M., Zhu, R., Lucey, S.: Learning depth from monocular videos using direct methods. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2022\u20132030 (2018)","DOI":"10.1109\/CVPR.2018.00216"},{"key":"3498_CR26","doi-asserted-by":"crossref","unstructured":"Yang, N., Stumberg, L.v., Wang, R., Cremers, D.: D3vo: Deep depth, deep pose and deep uncertainty for monocular visual odometry. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1281\u20131292 (2020)","DOI":"10.1109\/CVPR42600.2020.00136"},{"key":"3498_CR27","first-page":"7281","volume":"34","author":"Y Yuan","year":"2021","unstructured":"Yuan, Y., Fu, R., Huang, L., Lin, W., Zhang, C., Chen, X., Wang, J.: Hrformer: High-resolution vision transformer for dense predict. Adv. Neural Inform. Process. Syst. 34, 7281\u20137293 (2021)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"3498_CR28","doi-asserted-by":"crossref","unstructured":"Zhang, N., Nex, F., Vosselman, G., Kerle, N.: Lite-mono: A lightweight cnn and transformer architecture for self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18,537\u201318,546 (2023)","DOI":"10.1109\/CVPR52729.2023.01778"},{"key":"3498_CR29","doi-asserted-by":"publisher","first-page":"3251","DOI":"10.1109\/TIP.2022.3167307","volume":"31","author":"Y Zhang","year":"2022","unstructured":"Zhang, Y., Gong, M., Li, J., Zhang, M., Jiang, F., Zhao, H.: Self-supervised monocular depth estimation with multiscale perception. IEEE Transact. Image Process. 31, 3251\u20133266 (2022)","journal-title":"IEEE Transact. Image Process."},{"key":"3498_CR30","doi-asserted-by":"crossref","unstructured":"Zhao, C., Zhang, Y., Poggi, M., Tosi, F., Guo, X., Zhu, Z., Huang, G., Tang, Y., Mattoccia, S.: Monovit: Self-supervised monocular depth estimation with a vision transformer. In: International Conference on 3D Vision (2022)","DOI":"10.1109\/3DV57658.2022.00077"},{"key":"3498_CR31","unstructured":"Zhou, H., Greenwood, D., Taylor, S.: Self-supervised monocular depth estimation with internal feature fusion. In: British Machine Vision Conference (BMVC) (2021)"},{"key":"3498_CR32","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., Lowe, D.G.: Unsupervised learning of depth and ego-motion from video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1851\u20131858 (2017)","DOI":"10.1109\/CVPR.2017.700"},{"key":"3498_CR33","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Fan, X., Shi, P., Xin, Y.: R-msfm: recurrent multi-scale feature modulation for monocular depth estimating. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12,777\u201312,786 (2021)","DOI":"10.1109\/ICCV48922.2021.01254"},{"key":"3498_CR34","doi-asserted-by":"crossref","unstructured":"Zhu, S., Brazil, G., Liu, X.: The edge of depth: Explicit constraints between segmentation and depth. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13,116\u201313,125 (2020)","DOI":"10.1109\/CVPR42600.2020.01313"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03498-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-024-03498-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03498-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T13:01:39Z","timestamp":1732280499000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-024-03498-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,22]]},"references-count":34,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2024,7]]}},"alternative-id":["3498"],"URL":"https:\/\/doi.org\/10.1007\/s00371-024-03498-w","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6,22]]},"assertion":[{"value":"15 May 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 June 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}