{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T06:58:06Z","timestamp":1783407486619,"version":"3.54.6"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,5,25]],"date-time":"2021-05-25T00:00:00Z","timestamp":1621900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,5,25]],"date-time":"2021-05-25T00:00:00Z","timestamp":1621900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["41774027"],"award-info":[{"award-number":["41774027"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2242020R40135"],"award-info":[{"award-number":["2242020R40135"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Foundation of Key Laboratory of Micro-Inertial Instrument and Advanced Navigation Technology, Ministry of Education","award":["SEU-MIAN-201801"],"award-info":[{"award-number":["SEU-MIAN-201801"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2022,5]]},"DOI":"10.1007\/s00371-021-02092-8","type":"journal-article","created":{"date-parts":[[2021,5,25]],"date-time":"2021-05-25T08:02:52Z","timestamp":1621929772000},"page":"1619-1630","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Attention Unet++ for lightweight depth estimation from sparse depth samples and a single RGB image"],"prefix":"10.1007","volume":"38","author":[{"given":"Tao","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuguo","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Sheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingchun","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiansheng","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,5,25]]},"reference":[{"key":"2092_CR1","doi-asserted-by":"publisher","first-page":"1053","DOI":"10.1007\/s00371-019-01714-6","volume":"36","author":"M He","year":"2020","unstructured":"He, M., Zhu, C., Huang, Q., Ren, B., Liu, J.: A review of monocular visual odometry. Vis. Comput. 36, 1053\u20131065 (2020). https:\/\/doi.org\/10.1007\/s00371-019-01714-6","journal-title":"Vis. Comput."},{"key":"2092_CR2","doi-asserted-by":"publisher","first-page":"829","DOI":"10.1007\/s00371-018-1550-6","volume":"34","author":"G Bui","year":"2018","unstructured":"Bui, G., Le, T., Morago, B., Duan, Y.: Point-based rendering enhancement via deep learning. Vis. Comput. 34, 829\u2013841 (2018). https:\/\/doi.org\/10.1007\/s00371-018-1550-6","journal-title":"Vis. Comput."},{"key":"2092_CR3","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-020-01934-1","author":"Z Zhang","year":"2020","unstructured":"Zhang, Z., Lian, D., Gao, S.: RGB-D-based gaze point estimation via multi-column CNNs and facial landmarks global optimization. Vis. Comput. (2020). https:\/\/doi.org\/10.1007\/s00371-020-01934-1","journal-title":"Vis. Comput."},{"key":"2092_CR4","doi-asserted-by":"crossref","unstructured":"Wofk, D., Ma, F., Yang, T.-J., Karaman, S., Sze, V.: Fastdepth: fast monocular depth estimation on embedded systems. In: 2019 International Conference on Robotics and Automation (ICRA), pp. 6101\u20136108. IEEE (2019)","DOI":"10.1109\/ICRA.2019.8794182"},{"key":"2092_CR5","doi-asserted-by":"crossref","unstructured":"Xian, K., Shen, C., Cao, Z., Lu, H., Xiao, Y., Li, R., Luo, Z.: Monocular relative depth perception with web stereo data supervision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 311\u2013320 (2018)","DOI":"10.1109\/CVPR.2018.00040"},{"key":"2092_CR6","doi-asserted-by":"crossref","unstructured":"Laina, I., Rupprecht, C., Belagiannis, V., Tombari, F., Navab, N.: Deeper depth prediction with fully convolutional residual networks. In: 2016 Fourth International Conference on 3D Vision (3DV), pp. 239\u2013248. IEEE (2016)","DOI":"10.1109\/3DV.2016.32"},{"key":"2092_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-020-01832-6","author":"J Shi","year":"2020","unstructured":"Shi, J., Sun, Y., Bai, S., Sun, Z., Tian, Z.: A self-supervised method of single-image depth estimation by feeding forward information using max-pooling layers. Vis. Comput. (2020). https:\/\/doi.org\/10.1007\/s00371-020-01832-6","journal-title":"Vis. Comput."},{"key":"2092_CR8","doi-asserted-by":"publisher","first-page":"1165","DOI":"10.1007\/s00371-018-1551-5","volume":"34","author":"P Guerrero","year":"2018","unstructured":"Guerrero, P., Winnem\u00f6ller, H., Li, W., Mitra, N.J.: DepthCut: improved depth edge estimation using multiple unreliable channels. Vis. Comput. 34, 1165\u20131176 (2018). https:\/\/doi.org\/10.1007\/s00371-018-1551-5","journal-title":"Vis. Comput."},{"key":"2092_CR9","doi-asserted-by":"crossref","unstructured":"Mal, F., Karaman, S.: Sparse-to-dense: depth prediction from sparse depth samples and a single image. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 1\u20138. IEEE (2018)","DOI":"10.1109\/ICRA.2018.8460184"},{"key":"2092_CR10","doi-asserted-by":"publisher","first-page":"806","DOI":"10.1109\/TCI.2020.2981761","volume":"6","author":"P Hambarde","year":"2020","unstructured":"Hambarde, P., Murala, S.: S2DNet: depth estimation from single image and sparse samples. IEEE Trans. Comput. Imag. 6, 806\u2013817 (2020). https:\/\/doi.org\/10.1109\/TCI.2020.2981761","journal-title":"IEEE Trans. Comput. Imag."},{"key":"2092_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Z., Badrinarayanan, V., Drozdov, G., Rabinovich, A.: Estimating depth from rgb and sparse sensing. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 167\u2013182 (2018)","DOI":"10.1007\/978-3-030-01225-0_11"},{"key":"2092_CR12","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1007\/978-3-030-01270-0_7","volume-title":"Computer Vision\u2014ECCV 2018","author":"X Cheng","year":"2018","unstructured":"Cheng, X., Wang, P., Yang, R.: Depth estimation via affinity learned with convolutional spatial propagation network. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) Computer Vision\u2014ECCV 2018, pp. 108\u2013125. Springer International Publishing, Cham (2018)"},{"key":"2092_CR13","doi-asserted-by":"publisher","first-page":"10615","DOI":"10.1609\/aaai.v34i07.6635","volume":"34","author":"X Cheng","year":"2020","unstructured":"Cheng, X., Wang, P., Guan, C., Yang, R.: CSPN++: learning context and resource aware convolutional spatial propagation networks for depth completion. AAAI. 34, 10615\u201310622 (2020). https:\/\/doi.org\/10.1609\/aaai.v34i07.6635","journal-title":"AAAI."},{"key":"2092_CR14","doi-asserted-by":"crossref","unstructured":"Qiu, J., Cui, Z., Zhang, Y., Zhang, X., Liu, S., Zeng, B., Pollefeys, M.: DeepLiDAR: deep surface normal guided depth prediction for outdoor scene from sparse LiDAR data and single color image. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3308\u20133317. IEEE, Long Beach, CA, USA (2019)","DOI":"10.1109\/CVPR.2019.00343"},{"key":"2092_CR15","doi-asserted-by":"crossref","unstructured":"Shivakumar, S.S., Nguyen, T., Miller, I.D., Chen, S.W., Kumar, V., Taylor, C.J.: Dfusenet: deep fusion of RGB and sparse depth information for image guided dense depth completion. In: 2019 IEEE Intelligent Transportation Systems Conference (ITSC), pp. 13\u201320. IEEE (2019)","DOI":"10.1109\/ITSC.2019.8917294"},{"key":"2092_CR16","unstructured":"Tang, J., Tian, F.-P., Feng, W., Li, J., Tan, P.: Learning Guided Convolutional Network for Depth Completion. arXiv:1908.01238 [cs]. (2019)"},{"key":"2092_CR17","doi-asserted-by":"crossref","unstructured":"Hawe, S., Kleinsteuber, M., Diepold, K.: Dense disparity maps from sparse disparity measurements. In: 2011 International Conference on Computer Vision, pp. 2126\u20132133. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126488"},{"key":"2092_CR18","doi-asserted-by":"publisher","first-page":"1983","DOI":"10.1109\/TIP.2015.2409551","volume":"24","author":"L-K Liu","year":"2015","unstructured":"Liu, L.-K., Chan, S.H., Nguyen, T.Q.: Depth reconstruction from sparse samples: representation, algorithm, and sampling. IEEE Trans. Image Process. 24, 1983\u20131996 (2015)","journal-title":"IEEE Trans. Image Process."},{"key":"2092_CR19","doi-asserted-by":"crossref","unstructured":"Uhrig, J., Schneider, N., Schneider, L., Franke, U., Brox, T., Geiger, A.: Sparsity Invariant CNNs. In: 2017 International Conference on 3D Vision (3DV), pp. 11\u201320. IEEE, Qingdao (2017)","DOI":"10.1109\/3DV.2017.00012"},{"key":"2092_CR20","doi-asserted-by":"crossref","unstructured":"Ma, F., Cavalheiro, G.V., Karaman, S.: Self-supervised Sparse-to-Dense: Self-supervised Depth Completion from LiDAR and Monocular Camera. arXiv:1807.00275 [cs]. (2018)","DOI":"10.1109\/ICRA.2019.8793637"},{"key":"2092_CR21","unstructured":"Iandola, F.N., Han, S., Moskewicz, M.W., Ashraf, K., Dally, W.J., Keutzer, K.: SqueezeNet: AlexNet-Level Accuracy with 50x Fewer Parameters and< 0.5 MB Model Size. arXiv preprint arXiv:1602.07360. (2016)"},{"key":"2092_CR22","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Commun. ACM 60, 84\u201390 (2017)","journal-title":"Commun. ACM"},{"key":"2092_CR23","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: Efficient Convolutional Neural Networks for Mobile Vision Applications. arXiv preprint arXiv:1704.04861. (2017)"},{"key":"2092_CR24","unstructured":"Simonyan, K., Zisserman, A.: Very Deep Convolutional Networks for Large-Scale Image Recognition. arXiv preprint arXiv:1409.1556. (2014)"},{"key":"2092_CR25","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., Berg, A.C.: Ssd: single shot multibox detector. In: European Conference on Computer Vision, pp. 21\u201337. Springer (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"2092_CR26","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, pp. 91\u201399 (2015)"},{"key":"2092_CR27","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Rahman Siddiquee, M.M., Tajbakhsh, N., Liang, J.: UNet++: a nested U-net architecture for medical image segmentation. In: Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support, pp. 3\u201311. Springer International Publishing, Cham (2018)","DOI":"10.1007\/978-3-030-00889-5_1"},{"key":"2092_CR28","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"2092_CR29","doi-asserted-by":"crossref","unstructured":"Xie, S., Tu, Z.: Holistically-nested edge detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1395\u20131403 (2015)","DOI":"10.1109\/ICCV.2015.164"},{"key":"2092_CR30","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 234\u2013241. Springer (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2092_CR31","doi-asserted-by":"publisher","first-page":"9645","DOI":"10.1109\/TIM.2020.3005230","volume":"69","author":"H Li","year":"2020","unstructured":"Li, H., Wu, X.-J., Durrani, T.: NestFuse: an infrared and visible image fusion architecture based on nest connection and spatial\/channel attention models. IEEE Trans. Instrum. Meas. 69, 9645\u20139656 (2020). https:\/\/doi.org\/10.1109\/TIM.2020.3005230","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"2092_CR32","doi-asserted-by":"crossref","unstructured":"Huang, H., Lin, L., Tong, R., Hu, H., Zhang, Q., Iwamoto, Y., Han, X., Chen, Y.-W., Wu, J.: UNet 3+: a full-scale connected U-net for medical image segmentation. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1055\u20131059. IEEE, Barcelona, Spain (2020)","DOI":"10.1109\/ICASSP40776.2020.9053405"},{"key":"2092_CR33","doi-asserted-by":"crossref","unstructured":"Drozdzal, M., Vorontsov, E., Chartrand, G., Kadoury, S., Pal, C.: The Importance of Skip Connections in Biomedical Image Segmentation. arXiv:1608.04117 [cs]. (2016)","DOI":"10.1007\/978-3-319-46976-8_19"},{"key":"2092_CR34","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2092_CR35","doi-asserted-by":"crossref","unstructured":"Wang, F., Jiang, M., Qian, C., Yang, S., Li, C., Zhang, H., Wang, X., Tang, X.: Residual attention network for image classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3156\u20133164 (2017)","DOI":"10.1109\/CVPR.2017.683"},{"key":"2092_CR36","doi-asserted-by":"publisher","first-page":"1261","DOI":"10.1007\/s00371-019-01733-3","volume":"36","author":"J Cai","year":"2020","unstructured":"Cai, J., Hu, J.: 3D RANs: 3D residual attention networks for action recognition. Vis. Comput. 36, 1261\u20131270 (2020). https:\/\/doi.org\/10.1007\/s00371-019-01733-3","journal-title":"Vis. Comput."},{"key":"2092_CR37","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-020-02001-5","author":"Z Rao","year":"2020","unstructured":"Rao, Z., He, M., Dai, Y., Shen, Z.: Patch attention network with generative adversarial model for semi-supervised binocular disparity prediction. Vis. Comput. (2020). https:\/\/doi.org\/10.1007\/s00371-020-02001-5","journal-title":"Vis. Comput."},{"key":"2092_CR38","unstructured":"Hu, J., Shen, L., Albanie, S., Sun, G., Vedaldi, A.: Gather-excite: exploiting feature context in convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 9401\u20139411 (2018)"},{"key":"2092_CR39","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"2092_CR40","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K.: Non-local Neural Networks. arXiv:1711.07971 [cs]. (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"2092_CR41","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., So Kweon, I.: Cbam: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"2092_CR42","unstructured":"Zhang, H., Goodfellow, I., Metaxas, D., Odena, A.: Self-attention generative adversarial networks. In: International Conference on Machine Learning, pp. 7354\u20137363. PMLR (2019)"},{"key":"2092_CR43","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-020-01821-9","author":"Z Liu","year":"2020","unstructured":"Liu, Z., Duan, Q., Shi, S., Zhao, P.: Multi-level progressive parallel attention guided salient object detection for RGB-D images. Vis. Comput. (2020). https:\/\/doi.org\/10.1007\/s00371-020-01821-9","journal-title":"Vis. Comput."},{"key":"2092_CR44","unstructured":"Park, J., Woo, S., Lee, J.-Y., Kweon, I.S.: Bam: Bottleneck Attention Module. arXiv preprint arXiv:1807.06514. (2018)"},{"key":"2092_CR45","unstructured":"Tu, Z., Lee, C.-Y., Xie, S.: Deeply-supervised nets. Presented at the (2014)"},{"key":"2092_CR46","doi-asserted-by":"crossref","unstructured":"Yu, J., Lin, Z., Yang, J., Shen, X., Lu, X., Huang, T.S.: Free-form image inpainting with gated convolution. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4471\u20134480 (2019)","DOI":"10.1109\/ICCV.2019.00457"},{"key":"2092_CR47","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: European Conference on Computer Vision, pp. 746\u2013760. Springer (2012)","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"2092_CR48","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., Urtasun, R.: Are we ready for autonomous driving? The kitti vision benchmark suite. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition, pp. 3354\u20133361. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"2092_CR49","doi-asserted-by":"crossref","unstructured":"Liao, Y., Huang, L., Wang, Y., Kodagoda, S., Yu, Y., Liu, Y.: Parse geometry from a line: monocular depth estimation with partial laser observation. In: 2017 IEEE International Conference on Robotics and Automation (ICRA), pp. 5059\u20135066. IEEE (2017)","DOI":"10.1109\/ICRA.2017.7989590"},{"key":"2092_CR50","doi-asserted-by":"crossref","unstructured":"Fu, C., Mertz, C., Dolan, J.M.: LIDAR and monocular camera fusion: on-road depth completion for autonomous driving. In: 2019 IEEE Intelligent Transportation Systems Conference (ITSC), pp. 273\u2013278. IEEE (2019)","DOI":"10.1109\/ITSC.2019.8917201"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02092-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-021-02092-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-021-02092-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,13]],"date-time":"2022-04-13T13:12:17Z","timestamp":1649855537000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-021-02092-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,5,25]]},"references-count":50,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2022,5]]}},"alternative-id":["2092"],"URL":"https:\/\/doi.org\/10.1007\/s00371-021-02092-8","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,5,25]]},"assertion":[{"value":"10 February 2021","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 May 2021","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}