{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:28:09Z","timestamp":1740122889273,"version":"3.37.3"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"26","license":[{"start":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T00:00:00Z","timestamp":1706313600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T00:00:00Z","timestamp":1706313600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62006209"],"award-info":[{"award-number":["62006209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-18290-0","type":"journal-article","created":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T03:01:56Z","timestamp":1706324516000},"page":"68793-68811","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MDEConvFormer: estimating monocular depth as soft regression based on convolutional transformer"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6787-4384","authenticated-orcid":false,"given":"Wen","family":"Su","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ye","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haifeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenzhen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,27]]},"reference":[{"key":"18290_CR1","doi-asserted-by":"crossref","unstructured":"Han C, Cheng D, Kou Q, Wang X, Chen L, Zhao J (2022) Self-supervised monocular depth estimation with multi-scale structure similarity loss. Multimed Tools Appl pp 1\u201316","DOI":"10.1007\/s11042-022-14012-6"},{"key":"18290_CR2","doi-asserted-by":"crossref","unstructured":"Wang F-E, Yeh Y-H, Sun M, Chiu W-C, Tsai Y-H (2021) Led2-net: Monocular 360deg layout estimation via differentiable depth rendering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12956\u201312965","DOI":"10.1109\/CVPR46437.2021.01276"},{"issue":"14","key":"18290_CR3","doi-asserted-by":"publisher","first-page":"20771","DOI":"10.1007\/s11042-022-13921-w","volume":"82","author":"V-H Le","year":"2023","unstructured":"Le V-H (2023) Deep learning-based for human segmentation and tracking, 3d human pose estimation and action recognition on monocular video of mads dataset. Multimedia Tools and Applications 82(14):20771\u201320818","journal-title":"Multimedia Tools and Applications"},{"key":"18290_CR4","doi-asserted-by":"crossref","unstructured":"Hoyer L, Dai D, Chen Y, Koring A, Saha S, Van\u00a0Gool L (2021) Three ways to improve semantic segmentation with self-supervised depth estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11130\u201311140","DOI":"10.1109\/CVPR46437.2021.01098"},{"key":"18290_CR5","doi-asserted-by":"crossref","unstructured":"Zhu F, Liu L, Xie J, Shen F, Shao L, Fang Y (2018) Learning to synthesize 3d indoor scenes from monocular images. In: Proceedings of the 26th ACM international conference on multimedia, pp 501\u2013509","DOI":"10.1145\/3240508.3240700"},{"key":"18290_CR6","unstructured":"Chong Z, Ma X, Zhang H, Yue Y, Li H, Wang Z, Ouyang W (2022) Monodistill: Learning spatial features for monocular 3d object detection. arXiv preprint arXiv:2201.10830"},{"key":"18290_CR7","doi-asserted-by":"crossref","unstructured":"Tateno K, Tombari F, Laina I, Navab N (2017) Cnn-slam: Real-time dense monocular slam with learned depth prediction. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6243\u20136252","DOI":"10.1109\/CVPR.2017.695"},{"key":"18290_CR8","doi-asserted-by":"crossref","unstructured":"Hedau V, Hoiem D, Forsyth D (2010) Thinking inside the box: Using appearance models and context based on room geometry. In: Computer Vision\u2013ECCV 2010: 11th European Conference on Computer Vision, Heraklion, Crete, Greece, September 5-11, 2010, Proceedings, Part VI 11, Springer, pp 224\u2013237","DOI":"10.1007\/978-3-642-15567-3_17"},{"issue":"11","key":"18290_CR9","doi-asserted-by":"publisher","first-page":"2144","DOI":"10.1109\/TPAMI.2014.2316835","volume":"36","author":"K Karsch","year":"2014","unstructured":"Karsch K, Liu C, Kang SB (2014) Depth transfer: Depth extraction from video using non-parametric sampling. IEEE Trans Pattern Anal Mach Intell 36(11):2144\u20132158","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18290_CR10","unstructured":"Eigen D, Puhrsch C, Fergus R (2014) Depth map prediction from a single image using a multi-scale deep network. Adv Neural Inform Process Syst 27"},{"key":"18290_CR11","doi-asserted-by":"crossref","unstructured":"Eigen D, Fergus R (2015) Predicting depth, surface normals and semantic labels with a common multi-scale convolutional architecture. In: Proceedings of the IEEE Int Conf Comput Vis pp 2650\u20132658","DOI":"10.1109\/ICCV.2015.304"},{"key":"18290_CR12","doi-asserted-by":"crossref","unstructured":"Fu H, Gong M, Wang C, Batmanghelich K, Tao D (2018) Deep ordinal regression network for monocular depth estimation. In: Proceedings of the IEEE Conference on computer vision and pattern recognition, pp 2002\u20132011","DOI":"10.1109\/CVPR.2018.00214"},{"key":"18290_CR13","doi-asserted-by":"crossref","unstructured":"Yuan W, Gu X, Dai Z, Zhu S, Tan P (2022) New crfs: Neural window fully-connected crfs for monocular depth estimation. arXiv preprint arXiv:2203.01502","DOI":"10.1109\/CVPR52688.2022.00389"},{"key":"18290_CR14","doi-asserted-by":"crossref","unstructured":"Tomar SS, Suin M, Rajagopalan A (2022) Hybrid transformer based feature fusion for self-supervised monocular depth estimation. In: European conference on computer vision, Springer, pp 308\u2013326","DOI":"10.1007\/978-3-031-25063-7_19"},{"issue":"11","key":"18290_CR15","doi-asserted-by":"publisher","first-page":"3174","DOI":"10.1109\/TCSVT.2017.2740321","volume":"28","author":"Y Cao","year":"2017","unstructured":"Cao Y, Wu Z, Shen C (2017) Estimating depth from monocular images as classification using deep fully convolutional residual networks. IEEE Trans Circuits Syst Video Technol 28(11):3174\u20133182","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"18290_CR16","doi-asserted-by":"crossref","unstructured":"Laina I, Rupprecht C, Belagiannis V, Tombari F, Navab N (2016) Deeper depth prediction with fully convolutional residual networks. In: 2016 Fourth international conference on 3D vision (3DV), IEEE, pp 239\u2013248","DOI":"10.1109\/3DV.2016.32"},{"key":"18290_CR17","doi-asserted-by":"crossref","unstructured":"Jiao J, Cao Y, Song Y, Lau R (2018) Look deeper into depth: Monocular depth estimation with semantic booster and attention-driven loss. In: Proceedings of the European conference on computer vision (ECCV), pp 53\u201369","DOI":"10.1007\/978-3-030-01267-0_4"},{"key":"18290_CR18","doi-asserted-by":"crossref","unstructured":"Li Z, Snavely N (2018) Megadepth: Learning single-view depth prediction from internet photos. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2041\u20132050","DOI":"10.1109\/CVPR.2018.00218"},{"key":"18290_CR19","doi-asserted-by":"crossref","unstructured":"Xu D, Ricci E, Ouyang W, Wang X, Sebe N (2017) Multi-scale continuous crfs as sequential deep networks for monocular depth estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5354\u20135362","DOI":"10.1109\/CVPR.2017.25"},{"key":"18290_CR20","doi-asserted-by":"crossref","unstructured":"Xu D, Wang W, Tang H, Liu H, Sebe N, Ricci E (2018) Structured attention guided convolutional neural fields for monocular depth estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3917\u20133925","DOI":"10.1109\/CVPR.2018.00412"},{"issue":"8","key":"18290_CR21","doi-asserted-by":"publisher","first-page":"4131","DOI":"10.1109\/TIP.2018.2836318","volume":"27","author":"Y Kim","year":"2018","unstructured":"Kim Y, Jung H, Min D, Sohn K (2018) Deep monocular depth estimation via integration of global and local predictions. IEEE Trans Image Process 27(8):4131\u20134144","journal-title":"IEEE Trans Image Process"},{"key":"18290_CR22","doi-asserted-by":"crossref","unstructured":"Lee J-H, Kim C-S (2019) Monocular depth estimation using relative depth maps. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9729\u20139738","DOI":"10.1109\/CVPR.2019.00996"},{"key":"18290_CR23","doi-asserted-by":"crossref","unstructured":"Godard C, Mac\u00a0Aodha O, Firman M, Brostow GJ (2019) Digging into self-supervised monocular depth estimation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3828\u20133838","DOI":"10.1109\/ICCV.2019.00393"},{"key":"18290_CR24","doi-asserted-by":"crossref","unstructured":"Chen X, Chen X, Zha Z-J (2019) Structure-aware residual pyramid network for monocular depth estimation. arXiv preprint arXiv:1907.06023","DOI":"10.24963\/ijcai.2019\/98"},{"issue":"11","key":"18290_CR25","doi-asserted-by":"publisher","first-page":"4381","DOI":"10.1109\/TCSVT.2021.3049869","volume":"31","author":"M Song","year":"2021","unstructured":"Song M, Lim S, Kim W (2021) Monocular depth estimation using laplacian pyramid-based depth residuals. IEEE Trans Circuits Syst Video Technol 31(11):4381\u20134393","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"18290_CR26","unstructured":"Yang J, An L, Dixit A, Koo J, Park SI (2022) Depth estimation with simplified transformer. arXiv preprint arXiv:2204.13791"},{"key":"18290_CR27","doi-asserted-by":"crossref","unstructured":"Ranftl R, Bochkovskiy A, Koltun V (2021) Vision transformers for dense prediction. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 12179\u201312188","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"18290_CR28","unstructured":"Bhat SF, Alhashim I, Wonka P (2021) Adabins: Depth estimation using adaptive bins. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4009\u20134018"},{"key":"18290_CR29","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie E, Wang W, Yu Z, Anandkumar A, Alvarez JM, Luo P (2021) Segformer: Simple and efficient design for semantic segmentation with transformers. Adv Neural Inf Process Syst 34:12077\u201312090","journal-title":"Adv Neural Inf Process Syst"},{"key":"18290_CR30","doi-asserted-by":"crossref","unstructured":"Ma F, Karaman S (2018) Sparse-to-dense: Depth prediction from sparse depth samples and a single image. In: 2018 IEEE International conference on robotics and automation (ICRA), IEEE, pp 4796\u20134803","DOI":"10.1109\/ICRA.2018.8460184"},{"key":"18290_CR31","doi-asserted-by":"crossref","unstructured":"Zhang Y, Funkhouser T (2018) Deep depth completion of a single rgb-d image. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 175\u2013185","DOI":"10.1109\/CVPR.2018.00026"},{"key":"18290_CR32","doi-asserted-by":"crossref","unstructured":"Mousavian A, Pirsiavash H, Ko\u0161eck\u00e1 J (2016) Joint semantic segmentation and depth estimation with deep convolutional networks. In: 2016 Fourth international conference on 3D vision (3DV), IEEE, pp 611\u2013619","DOI":"10.1109\/3DV.2016.69"},{"key":"18290_CR33","doi-asserted-by":"crossref","unstructured":"Kim S, Park K, Sohn K, Lin S (2016) Unified depth prediction and intrinsic image decomposition from a single image via joint convolutional neural fields. In: Computer vision\u2013ECCV 2016: 14th European conference, Amsterdam, The Netherlands, Proceedings, Part VIII 14, Springer, pp 143\u2013159. Accessed 11\u201314 Oct 2016","DOI":"10.1007\/978-3-319-46484-8_9"},{"key":"18290_CR34","doi-asserted-by":"crossref","unstructured":"Hu J, Ozay M, Zhang Y, Okatani T (2019) Revisiting single image depth estimation: Toward higher resolution maps with accurate object boundaries. In: 2019 IEEE Winter conference on applications of computer vision (WACV), IEEE, pp 1043\u20131051","DOI":"10.1109\/WACV.2019.00116"},{"key":"18290_CR35","doi-asserted-by":"crossref","unstructured":"Kusupati U, Cheng S, Chen R, Su H (2020) Normal assisted stereo depth estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2189\u20132199","DOI":"10.1109\/CVPR42600.2020.00226"},{"key":"18290_CR36","doi-asserted-by":"crossref","unstructured":"Wang P, Shen X, Lin Z, Cohen S, Price B, Yuille AL (2015) Towards unified depth and semantic prediction from a single image. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2800\u20132809","DOI":"10.1109\/CVPR.2015.7298897"},{"key":"18290_CR37","doi-asserted-by":"crossref","unstructured":"Lee J-H, Kim C-S (2020) Multi-loss rebalancing algorithm for monocular depth estimation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, Proceedings, Part XVII 16, Springer, pp 785\u2013801. Accessed 23\u201328 Aug 2020","DOI":"10.1007\/978-3-030-58520-4_46"},{"key":"18290_CR38","doi-asserted-by":"crossref","unstructured":"Godard C, Mac\u00a0Aodha O, Brostow GJ (2017) Unsupervised monocular depth estimation with left-right consistency. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 270\u2013279","DOI":"10.1109\/CVPR.2017.699"},{"key":"18290_CR39","doi-asserted-by":"crossref","unstructured":"Yu Z, Jin L, Gao S (2020) P 2 net: Patch-match and plane-regularization for unsupervised indoor depth estimation. In: European conference on computer vision, Springer, pp 206\u2013222","DOI":"10.1007\/978-3-030-58586-0_13"},{"key":"18290_CR40","doi-asserted-by":"crossref","unstructured":"Wang L, Zhang J, Wang, Y Lu H, Ruan X (2020) Cliffnet for monocular depth estimation with hierarchical embedding loss. In: European Conference on Computer Vision, Springer, pp 316\u2013331","DOI":"10.1007\/978-3-030-58558-7_19"},{"issue":"3","key":"18290_CR41","doi-asserted-by":"publisher","first-page":"1623","DOI":"10.1109\/TPAMI.2020.3019967","volume":"44","author":"R Ranftl","year":"2022","unstructured":"Ranftl R, Lasinger K, Hafner D, Schindler K, Koltun V (2022) Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer. IEEE Trans Pattern Anal Mach Intell 44(3):1623\u20131637","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18290_CR42","doi-asserted-by":"crossref","unstructured":"Shi W, Caballero J, Husz\u00e1r F, Totz J, Aitken AP, Bishop R, Rueckert D, Wang Z (2016) Real-time single image and video super-resolution using an efficient sub-pixel convolutional neural network. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1874\u20131883","DOI":"10.1109\/CVPR.2016.207"},{"key":"18290_CR43","doi-asserted-by":"crossref","unstructured":"Bae J, Moon S, Im S (2023) Deep digging into the generalization of self-supervised monocular depth estimation. In: Proceedings of the AAAI conference on artificial intelligence, vol 37, pp 187\u2013196","DOI":"10.1609\/aaai.v37i1.25090"},{"key":"18290_CR44","doi-asserted-by":"crossref","unstructured":"Silberman N, Hoiem D, Kohli P, Fergus R (2012) Indoor segmentation and support inference from rgbd images. In: Computer vision\u2013ECCV 2012: 12th European conference on computer vision, Florence, Italy, Proceedings, Part V 12, Springer, pp 746\u2013760. Accessed 7\u201313 Oct 2012","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"18290_CR45","doi-asserted-by":"crossref","unstructured":"Geiger A, Lenz P, Urtasun R (2012) Are we ready for autonomous driving? the kitti vision benchmark suite. In: 2012 IEEE Conference on computer vision and pattern recognition, IEEE, pp 3354\u20133361","DOI":"10.1109\/CVPR.2012.6248074"},{"issue":"10","key":"18290_CR46","doi-asserted-by":"publisher","first-page":"2024","DOI":"10.1109\/TPAMI.2015.2505283","volume":"38","author":"F Liu","year":"2015","unstructured":"Liu F, Shen C, Lin G, Reid I (2015) Learning depth from single monocular images using deep convolutional neural fields. IEEE Trans Pattern Anal Mach Intell 38(10):2024\u20132039","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18290_CR47","doi-asserted-by":"crossref","unstructured":"Xu X, Qiu J, Wang X, Wang Z (2022) Relationship spatialization for depth estimation. In: European conference on computer vision, Springer, pp 615\u2013637","DOI":"10.1007\/978-3-031-19836-6_35"},{"key":"18290_CR48","doi-asserted-by":"crossref","unstructured":"Pilzer A, Lathuiliere S, Sebe N, Ricci E (2019) Refine and distill: Exploiting cycle-inconsistency and knowledge distillation for unsupervised monocular depth estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9768\u20139777","DOI":"10.1109\/CVPR.2019.01000"},{"key":"18290_CR49","unstructured":"Alhashim I, Wonka P (2018) High quality monocular depth estimation via transfer learning. arXiv preprint arXiv:1812.11941"},{"key":"18290_CR50","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"issue":"8","key":"18290_CR51","doi-asserted-by":"publisher","first-page":"4009","DOI":"10.1007\/s11760-023-02631-x","volume":"17","author":"MK Kelishadrokhi","year":"2023","unstructured":"Kelishadrokhi MK, Ghattaei M, Fekri-Ershad S (2023) Innovative local texture descriptor in joint of human-based color features for content-based image retrieval. SIViP 17(8):4009\u20134017","journal-title":"SIViP"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18290-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-18290-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18290-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T01:15:54Z","timestamp":1721610954000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-18290-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,27]]},"references-count":51,"journal-issue":{"issue":"26","published-online":{"date-parts":[[2024,8]]}},"alternative-id":["18290"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-18290-0","relation":{},"ISSN":["1573-7721"],"issn-type":[{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2024,1,27]]},"assertion":[{"value":"18 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 December 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 January 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 January 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Text","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"The authors declare that they have no conflict of interest."}}]}}