{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T03:57:17Z","timestamp":1781150237660,"version":"3.54.1"},"reference-count":95,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.neunet.2026.108943","type":"journal-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T01:13:06Z","timestamp":1775437986000},"page":"108943","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Boosting self-supervised multi-frame depth estimation with hybrid geometric-semantic constraints"],"prefix":"10.1016","volume":"201","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1768-7465","authenticated-orcid":false,"given":"Jiaojiao","family":"Fang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"9","key":"10.1016\/j.neunet.2026.108943_bib0001","doi-asserted-by":"crossref","first-page":"2548","DOI":"10.1007\/s11263-021-01484-6","article-title":"Unsupervised scale-consistent depth learning from video","volume":"129","author":"Bian","year":"2021","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.neunet.2026.108943_bib0002","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","article-title":"Seeing through fog without Seeing fog: Deep multimodal sensor fusion in unseen adverse weather","author":"Bijelic","year":"2020"},{"key":"10.1016\/j.neunet.2026.108943_bib0003","article-title":"Pyramid stereo matching network","author":"Chang","year":"2018","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0004","unstructured":"J. Chen, Z. Xie, H. Liu, X. Liu, L. Tu, L. Li, and P. Luo, \u201cDDP: Diffusion model for dense visual prediction,\u201d in Proc. IEEE\/CVF Int. Conf. Comput. Vis., Paris, France, 2023, pp. 21741\u201321752."},{"issue":"3","key":"10.1016\/j.neunet.2026.108943_bib0005","doi-asserted-by":"crossref","first-page":"1328","DOI":"10.1109\/TCSVT.2021.3068834","article-title":"Fixing defect of photometric loss for self-supervised monocular depth estimation","volume":"32","author":"Chen","year":"2022","journal-title":"IEEE Transactions on Circuits and System for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0006","article-title":"Single-image depth perception in the wild","author":"Chen","year":"2016","journal-title":"In NeurIPS"},{"key":"10.1016\/j.neunet.2026.108943_bib0007","article-title":"Learning depth with convolutional spatial propagation network","author":"Cheng","year":"2019","journal-title":"PAMI"},{"key":"10.1016\/j.neunet.2026.108943_bib0008","article-title":"The cityscapes dataset for semantic urban scene understanding","author":"Cordts","year":"2016","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0009","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit","first-page":"5828","article-title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","author":"Dai","year":"2017"},{"key":"10.1016\/j.neunet.2026.108943_bib0010","article-title":"MVS2: Deep unsupervised multi-view stereo with multi-view symmetry","author":"Dai","year":"2019","journal-title":"In 3DV,"},{"key":"10.1016\/j.neunet.2026.108943_bib0011","series-title":"2024 IEEE International Conference on Robotics and Automation (ICRA)","first-page":"7318","article-title":"Mal: Motion-aware loss with temporal and distillation hints for self-supervised depth estimation","author":"Dong","year":"2024"},{"key":"10.1016\/j.neunet.2026.108943_bib0012","article-title":"Depth map prediction from a single image using a multi-scale deep network","author":"Eigen","year":"2014","journal-title":"In NeurIPS"},{"key":"10.1016\/j.neunet.2026.108943_bib0013","doi-asserted-by":"crossref","DOI":"10.1109\/LRA.2017.2715400","article-title":"Single-view and multi-view depth fusion","author":"Facil","year":"2017","journal-title":"IEEE Robotics and Automation Letters"},{"issue":"1","key":"10.1016\/j.neunet.2026.108943_bib0014","doi-asserted-by":"crossref","first-page":"329","DOI":"10.1109\/TCSVT.2023.3284479","article-title":"Iterdepth: Iterative residual refinement for outdoor self-supervised multi-frame monocular depth estimation","volume":"34","author":"Feng","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0015","doi-asserted-by":"crossref","DOI":"10.1109\/TIP.2025.3533207","article-title":"Self-supervised monocular depth estimation with dual-path encoders and offset field interpolation","volume":"34","author":"Feng","year":"2025","journal-title":"IEEE Transactions On Image Processing : A Publication of the IEEE Signal Processing Society"},{"key":"10.1016\/j.neunet.2026.108943_bib0016","series-title":"Proc. Eur. Conf. Comput. Vis","first-page":"228","article-title":"Disentangling object motion and occlusion for unsupervised multi-frame monocular depth","author":"Feng","year":"2022"},{"key":"10.1016\/j.neunet.2026.108943_bib0017","article-title":"Deep ordinal regression network for monocular depth estimation","author":"Fu","year":"2018","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0018","article-title":"Unsupervised CNN for single view depth estimation: Geometry to the rescue","author":"Garg","year":"2016","journal-title":"In ECCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0019","article-title":"Are we ready for autonomous driving? The KITTI Vision benchmark suite","author":"Geiger","year":"2012","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0020","article-title":"Unsupervised monocular depth estimation with left-right consistency","author":"Godard","year":"2017","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0021","article-title":"Digging into self-supervised monocular depth estimation","author":"Godard","year":"2019","journal-title":"In ICCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0022","article-title":"Forget about the LiDAR: Self-supervised depth estimators with MED probability volumes","author":"Gonzalez","year":"2020","journal-title":"NeurIPS"},{"key":"10.1016\/j.neunet.2026.108943_bib0023","article-title":"Depth from videos in the wild: Unsupervised monocular depth learning from unknown cameras","author":"Gordon","year":"2019","journal-title":"In ICCV"},{"issue":"5","key":"10.1016\/j.neunet.2026.108943_bib0024","doi-asserted-by":"crossref","first-page":"2844","DOI":"10.1109\/LRA.2023.3260724","article-title":"DRO: Deep recurrent optimizer for video to depth","volume":"8","author":"Gu","year":"2023","journal-title":"IEEE Robotics and Automation Letters"},{"key":"10.1016\/j.neunet.2026.108943_bib0025","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"160","article-title":"Multi-frame self-supervised depth with transformers","author":"Guizilini","year":"2022"},{"key":"10.1016\/j.neunet.2026.108943_bib0026","article-title":"Allan Raventos, and Adrien Gaidon. 3D packing for self-supervised monocular depth estimation","author":"Guizilini","year":"2020","journal-title":"In CVPR"},{"issue":"12","key":"10.1016\/j.neunet.2026.108943_bib0027","doi-asserted-by":"crossref","DOI":"10.3390\/rs16122221","article-title":"SIM-MultiDepth: Self-supervised indoor monocular multi-frame depth estimation based on texture-aware masking","volume":"16","author":"Guo","year":"2024","journal-title":"Remote Sensing"},{"issue":"133","key":"10.1016\/j.neunet.2026.108943_bib0028","article-title":"F\u00b2Depth: Self-supervised indoor monocular depth estimation via optical flow consistency and feature map synthesis","volume":"124","author":"Guo","year":"2024","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.neunet.2026.108943_bib0029","article-title":"Deep residual learning for image recognition","author":"He","year":"2016","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0030","article-title":"Multi-view stereo by temporal nonparametric fusion","author":"Hou","year":"2019","journal-title":"In ICCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0031","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"2005","article-title":"Depthcrafter: Generating consistent long depth sequences for open-world videos[C]","author":"Hu","year":"2025"},{"key":"10.1016\/j.neunet.2026.108943_bib0032","series-title":"2021 IEEE international conference on image processing (ICIP)","article-title":"M3VSNet: Unsupervised multi-metric multi-view stereo network","author":"Huang","year":"2021"},{"key":"10.1016\/j.neunet.2026.108943_bib0033","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"1665","article-title":"RM-depth: Unsupervised learning of recurrent monocular depth in dynamic scenes","author":"Hui","year":"2022"},{"key":"10.1016\/j.neunet.2026.108943_bib0034","article-title":"DPSNet: End-to-end deep plane sweep stereo","author":"Im","year":"2019","journal-title":"ICLR"},{"key":"10.1016\/j.neunet.2026.108943_bib0035","article-title":"Self-supervised monocular trained depth estimation using self-attention and discrete disparity volume","author":"Johnston","year":"2020","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0036","article-title":"Learning a multi-view stereo machine","author":"Kar","year":"2017","journal-title":"In NeurIPS"},{"key":"10.1016\/j.neunet.2026.108943_bib0037","article-title":"End-to-end learning of geometry and context for deep stereo regression","author":"Kendall","year":"2017","journal-title":"In ICCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0038","author":"Kingma","year":"2014","journal-title":"A method for stochastic optimization"},{"key":"10.1016\/j.neunet.2026.108943_bib0039","article-title":"Self-supervised monocular depth estimation: Solving the dynamic object problem by semantic guidance","author":"Klingner","year":"2020","journal-title":"In ECCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0040","article-title":"Supervising the new with the old: Learning SFM from SFM","author":"Klodt","year":"2018","journal-title":"In ECCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0041","series-title":"Proc. Eur. Conf. Comput. Vis. (ECCV)","article-title":"Indoor segmentation and support inference from rgbd images","author":"Kohli","year":"2012"},{"key":"10.1016\/j.neunet.2026.108943_bib0042","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"4842","article-title":"Attentive and contrastive learning for joint depth and motion field estimation","author":"Lee","year":"2021"},{"key":"10.1016\/j.neunet.2026.108943_bib0043","article-title":"Unsupervised monocular depth learning in dynamic scenes","author":"Li","year":"2020","journal-title":"In CoRL"},{"issue":"2","key":"10.1016\/j.neunet.2026.108943_bib0044","doi-asserted-by":"crossref","first-page":"830","DOI":"10.1109\/TCSVT.2022.3207105","article-title":"MonoIndoor++: Towards better practice of self-supervised monocular depth estimation for indoor environments","volume":"33","author":"Li","year":"2023","journal-title":"IEEE Transactions on Circuits and System for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0045","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"Learning to fuse monocular and multi-view cues for multi-frame depth estimation in dynamic scenes","author":"Li","year":"2023"},{"key":"10.1016\/j.neunet.2026.108943_bib0046","article-title":"Learning for disparity estimation through feature constancy","author":"Liang","year":"2018","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0047","series-title":"European conference on computer vision","first-page":"90","article-title":"Mono-ViFI: A unified learning framework for self-supervised single and multi-frame monocular depth estimation","author":"Liu","year":"2025"},{"issue":"12","key":"10.1016\/j.neunet.2026.108943_bib0048","doi-asserted-by":"crossref","first-page":"7565","DOI":"10.1109\/TCSVT.2023.3275584","article-title":"Self-supervised monocular depth estimation with self-reference distillation and disparity offset refinement","volume":"33","author":"Liu","year":"2023","journal-title":"IEEE Transaction on Circuits and System for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0049","article-title":"Occlusion-aware depth estimation with adaptive normal constraints","author":"Long","year":"2020","journal-title":"In ECCV"},{"issue":"4","key":"10.1016\/j.neunet.2026.108943_bib0050","doi-asserted-by":"crossref","first-page":"12291","DOI":"10.1109\/LRA.2022.3214787","article-title":"Two-stream based multi-stage hybrid decoder for self-supervised multi-frame monocular depth","volume":"7","author":"Long","year":"2022","journal-title":"IEEE Robotics and Automation Letters"},{"issue":"10","key":"10.1016\/j.neunet.2026.108943_bib0051","doi-asserted-by":"crossref","first-page":"2624","DOI":"10.1109\/TPAMI.2019.2930258","article-title":"Every pixel counts ++: Joint learning of geometry and motion with 3D holistic understanding","volume":"42","author":"Luo","year":"2020","journal-title":"IEEE Transactions Pattern Analysis Mach. Intelligent"},{"key":"10.1016\/j.neunet.2026.108943_bib0052","doi-asserted-by":"crossref","DOI":"10.1145\/3386569.3392377","article-title":"Consistent video depth estimation","author":"Luo","year":"2020","journal-title":"In ACM SIGGRAPH"},{"key":"10.1016\/j.neunet.2026.108943_bib0053","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit.","first-page":"5667","article-title":"Unsupervised learning of depth and ego-motion from monocular video using 3D geometric constraints","author":"Mahjourian","year":"2018"},{"key":"10.1016\/j.neunet.2026.108943_bib0054","article-title":"Fusion of stereo and still monocular depth estimates in a self-supervised learning context","author":"Martins","year":"2018","journal-title":"In ICRA"},{"key":"10.1016\/j.neunet.2026.108943_bib0055","article-title":"Ds-depth: Dynamic and static depth estimation via a fusion cost volume","author":"Miao","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0056","article-title":"Kinectfusion: Real-time dense surface mapping and tracking","author":"Newcombe","year":"2011","journal-title":"In UIST"},{"key":"10.1016\/j.neunet.2026.108943_bib0057","doi-asserted-by":"crossref","unstructured":"V. Patil, C. Sakaridis, A. Liniger, and L. Van Gool, \u201cP3Depth: Monocular depth estimation with a piecewise planarity prior,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., New Orleans, LA, USA, 2022, pp. 1610\u20131621.","DOI":"10.1109\/CVPR52688.2022.00166"},{"key":"10.1016\/j.neunet.2026.108943_bib0058","doi-asserted-by":"crossref","DOI":"10.1109\/LRA.2020.3017478","article-title":"Don\u2019t forget the past: Recurrent depth estimation from monocular video","author":"Patil","year":"2020","journal-title":"In IEEE Robotics and Automation Letters"},{"key":"10.1016\/j.neunet.2026.108943_bib0059","series-title":"Proc. Int. Conf. 3D Vis. (3DV)","first-page":"837","article-title":"Attention meets geometry: Geometry guided spatial-temporal attention for consistent self-supervised monocular depth estimation","author":"Ruhkamp","year":"2021"},{"key":"10.1016\/j.neunet.2026.108943_bib0060","doi-asserted-by":"crossref","DOI":"10.1007\/s11263-015-0816-y","article-title":"Imagenet large scale visual recognition challenge","author":"Russakovsky","year":"2015","journal-title":"IJCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0061","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"8907","article-title":"Self-supervised monocular depth estimation: Let's talk about the weather","author":"Saunders","year":"2023"},{"key":"10.1016\/j.neunet.2026.108943_bib0062","doi-asserted-by":"crossref","DOI":"10.1109\/ISCAIT64916.2025.11010448","article-title":"Pose-depth joint optimization and adaptive sampling for self-supervised multi-frame depth estimation","author":"Shen","year":"2025","journal-title":"2025 4th Internatinal. Symp. Computer Applied Information and Technology (ISCAIT)"},{"key":"10.1016\/j.neunet.2026.108943_bib0063","article-title":"Feature-metric loss for self-supervised learning of depth and ego-motion","author":"Shu","year":"2020","journal-title":"In ECCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0064","article-title":"On the importance of stereo for accurate depth estimation: An efficient semi-supervised deep neural network approach","author":"Smolyanskiy","year":"2018","journal-title":"In CVPR Workshops"},{"key":"10.1016\/j.neunet.2026.108943_bib0065","doi-asserted-by":"crossref","first-page":"3517","DOI":"10.1109\/TMM.2023.3312950","article-title":"Unsupervised monocular estimation of depth and visual odometry using attention and depth-pose consistency loss","volume":"26","author":"Song","year":"2023","journal-title":"IEEE Transaction Multimedia"},{"key":"10.1016\/j.neunet.2026.108943_bib0066","series-title":"Proc. Int. Conf. 3D Vis. (3DV)","first-page":"11","article-title":"Sparsity invariant CNNs","author":"Uhrig","year":"2017"},{"key":"10.1016\/j.neunet.2026.108943_bib0067","article-title":"DeMoN: Depth and motion network for learning monocular stereo","author":"Ummenhofer","year":"2017","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0068","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"12889","article-title":"Scaling local self-attention for parameter efficient visual backbones","author":"Vaswani","year":"2021"},{"key":"10.1016\/j.neunet.2026.108943_bib0069","series-title":"2024 IEEE International Conference on Robotics and Automation (ICRA)","first-page":"4976","article-title":"Weatherdepth: Curriculum contrastive learning for self-supervised depth estimation under adverse weather conditions","author":"Wang","year":"2024"},{"key":"10.1016\/j.neunet.2026.108943_bib0070","unstructured":"J. Wang, G. Zhang, Z. Wu, X. Li, and L. Liu, \u201cSelf-supervised joint learning framework of depth estimation via implicit cues,\u201d 2020, arXiv:2006.09876."},{"key":"10.1016\/j.neunet.2026.108943_bib0071","article-title":"Recurrent neural network for (un-)supervised learning of monocular video visual odometry and depth","author":"Wang","year":"2019","journal-title":"In CVPR"},{"issue":"3","key":"10.1016\/j.neunet.2026.108943_bib0072","doi-asserted-by":"crossref","first-page":"2689","DOI":"10.1609\/aaai.v37i3.25368","article-title":"Crafting monocular cues and velocity guidance for self-supervised multi-frame depth learning","volume":"37","author":"Wang","year":"2023","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"issue":"6","key":"10.1016\/j.neunet.2026.108943_bib0073","doi-asserted-by":"crossref","first-page":"5713","DOI":"10.1609\/aaai.v38i6.28383","article-title":"Sqldepth: Generalizable self-supervised fine-structured monocular depth estimation","volume":"38","author":"Wang","year":"2024","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"10.1016\/j.neunet.2026.108943_bib0074","article-title":"UnOS: Unified unsupervised optical-flow and stereo-depth estimation by watching videos","author":"Wang","year":"2019","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0075","article-title":"The temporal opportunist: Self-supervised multi-frame monocular depth","author":"Watson","year":"2021","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"10.1016\/j.neunet.2026.108943_bib0076","article-title":"Self-supervised monocular depth hints","author":"Watson","year":"2019","journal-title":"In ICCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0077","series-title":"European Conference on Computer Vision","first-page":"201","article-title":"ProDepth: Boosting self-supervised multi-frame monocular depth with probabilistic fusion[C]","author":"Woo","year":"2024"},{"issue":"6","key":"10.1016\/j.neunet.2026.108943_bib0078","doi-asserted-by":"crossref","first-page":"4989","DOI":"10.1109\/TCSVT.2023.3340948","article-title":"Self-supervised multi-frame monocular depth estimation for dynamic scenes","volume":"34","author":"Wu","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.108943_bib0079","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"20207","article-title":"GoMVS: Geometrically consistent cost aggregation for multi-view stereo","author":"Wu","year":"2024"},{"key":"10.1016\/j.neunet.2026.108943_bib0080","doi-asserted-by":"crossref","first-page":"22568","DOI":"10.1109\/TASE.2025.3619093","article-title":"Multi-view stereo with geometric encoding for large-scale dense scene reconstruction","volume":"22","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Automation Science and Engineering"},{"key":"10.1016\/j.neunet.2026.108943_bib0081","first-page":"1","article-title":"Toward end-to-end underwater multi-view stereo for real-world dense scene reconstruction","volume":"63","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.neunet.2026.108943_bib0082","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"10371","article-title":"Depth anything: Unleashing the power of large-scale unlabeled data","author":"Yang","year":"2024"},{"key":"10.1016\/j.neunet.2026.108943_bib0083","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"767","article-title":"Mvsnet: Depth inference for unstructured multi-view stereo","author":"Yao","year":"2018"},{"key":"10.1016\/j.neunet.2026.108943_bib0084","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9043","article-title":"Metric3d: Towards zero-shot metric 3d prediction from a single image[C]","author":"Yin","year":"2023"},{"key":"10.1016\/j.neunet.2026.108943_bib0085","article-title":"Towards accurate reconstruction of 3d scene shape from a single monocular image","author":"Yin","year":"2022","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.108943_bib0086","first-page":"1983","article-title":"GeoNet: Unsupervised learning of dense depth, optical flow and camera pose","author":"Yin","year":"2018","journal-title":"In Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.neunet.2026.108943_bib0087","doi-asserted-by":"crossref","unstructured":"Y. Yu, Z. Zheng, D. Lian, Z. Zhou, and S. Gao, \u201cSingle-image piecewise planar 3D reconstruction via associative embedding,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., Long Beach, CA, USA, 2019, pp. 1029\u20131037.","DOI":"10.1109\/CVPR.2019.00112"},{"key":"10.1016\/j.neunet.2026.108943_bib0088","article-title":"Unsupervised learning of monocular depth estimation and visual odometry with deep feature reconstruction","author":"Zhan","year":"2018","journal-title":"In CVPR"},{"key":"10.1016\/j.neunet.2026.108943_bib0089","article-title":"Domain-invariant stereo matching networks","author":"Zhang","year":"2020","journal-title":"In ECCV"},{"key":"10.1016\/j.neunet.2026.108943_bib0090","article-title":"Exploiting temporal consistency for real-time video depth estimation","author":"Zhang","year":"2019","journal-title":"In ICCV"},{"issue":"12","key":"10.1016\/j.neunet.2026.108943_bib0091","doi-asserted-by":"crossref","first-page":"17292","DOI":"10.1109\/TNNLS.2023.3301711","article-title":"Self-supervised monocular depth estimation with Self-perceptual anomaly handling","volume":"35","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.108943_bib0092","doi-asserted-by":"crossref","first-page":"3251","DOI":"10.1109\/TIP.2022.3167307","article-title":"Self-supervised monocular depth estimation with multiscale perception","volume":"31","author":"Zhang","year":"2022","journal-title":"IEEE Transactions On Image Processing : A Publication of the IEEE Signal Processing Society"},{"key":"10.1016\/j.neunet.2026.108943_bib0093","series-title":"Proc. Int. Conf. 3D Vis. (3DV)","first-page":"668","article-title":"MonoViT: Self-supervised monocular depth estimation with a vision transformer","author":"Zhao","year":"2022"},{"key":"10.1016\/j.neunet.2026.108943_bib0094","doi-asserted-by":"crossref","unstructured":"H. Zhao, S. Liu, S. Fu, Y. Liu, and T. Yu, \u201cTowards better generalization: Joint depth-pose learning without Posenet,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., Seattle, WA, USA, 2020, pp. 9151\u20139161.","DOI":"10.1109\/CVPR42600.2020.00917"},{"key":"10.1016\/j.neunet.2026.108943_bib0095","article-title":"Unsupervised learning of depth and ego-motion from video","author":"Zhou","year":"2017","journal-title":"In CVPR"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004041?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004041?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T03:06:30Z","timestamp":1781147190000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026004041"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":95,"alternative-id":["S0893608026004041"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108943","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Boosting self-supervised multi-frame depth estimation with hybrid geometric-semantic constraints","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.108943","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"108943"}}