{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:05:57Z","timestamp":1784203557838,"version":"3.55.0"},"reference-count":59,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neunet.2026.109049","type":"journal-article","created":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T23:29:15Z","timestamp":1778282955000},"page":"109049","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["VGLD: Visually-guided language disambiguation for monocular depth scale recovery"],"prefix":"10.1016","volume":"202","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-4550-841X","authenticated-orcid":false,"given":"Bojin","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5206-1110","authenticated-orcid":false,"given":"Jing","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109049_bib0001","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2039","article-title":"Learning to prompt clip for monocular depth estimation: Exploring the limits of human language","author":"Auty","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0002","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4009","article-title":"AdaBins: Depth estimation using adaptive bins","author":"Bhat","year":"2021"},{"key":"10.1016\/j.neunet.2026.109049_bib0003","series-title":"European conference on computer vision","first-page":"480","article-title":"LocalBins: Improving depth estimation by learning local distributions","author":"Bhat","year":"2022"},{"key":"10.1016\/j.neunet.2026.109049_bib0004","unstructured":"Bhat, S. F., Birkl, R., Wofk, D., Wonka, P., & M\u00fcller, M. (2023). ZoeDepth: Zero-shot transfer by combining relative and metric depth. arxiv: 2302.12288."},{"issue":"6","key":"10.1016\/j.neunet.2026.109049_bib0005","doi-asserted-by":"crossref","first-page":"380","DOI":"10.1007\/s00530-024-01590-8","article-title":"Hierarchical and progressive learning with key point sensitive loss for sonar image classification","volume":"30","author":"Chen","year":"2024","journal-title":"Multimedia Systems"},{"key":"10.1016\/j.neunet.2026.109049_bib0006","unstructured":"Cho, J., Min, D., Kim, Y., & Sohn, K. (2021). DIML\/CVL RGB-D dataset: 2M RGB-D images of natural indoor and outdoor scenes. arxiv: 2110.11590."},{"key":"10.1016\/j.neunet.2026.109049_bib0007","unstructured":"Cui, B., Huang, Y., Bai, L., & Ren, H. (2025). TR2M: Transferring monocular relative depth to metric depth with language descriptions and scale-oriented contrast. arXiv: 2506.13387."},{"key":"10.1016\/j.neunet.2026.109049_bib0008","series-title":"Advances in neural information processing systems","article-title":"Depth map prediction from a single image using a multi-scale deep network","volume":"vol. 27","author":"Eigen","year":"2014"},{"key":"10.1016\/j.neunet.2026.109049_bib0009","series-title":"European conference on computer vision","first-page":"241","article-title":"GeoWizard: Unleashing the diffusion priors for 3D geometry estimation from a single image","author":"Fu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0010","unstructured":"Ganj, A., Zhao, Y., Su, H., & Guo, T. (2023). Mobile AR depth estimation: Challenges & prospects\u2013extended version. In arxiv: 2310.14437."},{"key":"10.1016\/j.neunet.2026.109049_bib0011","series-title":"The KITTI vision benchmark suite. inCVPR","first-page":"5","article-title":"Are we ready forautonomous driving","volume":"vol. 2","author":"Geiger","year":"2012"},{"key":"10.1016\/j.neunet.2026.109049_bib0012","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2485","article-title":"3D packing for self-supervised monocular depth estimation","author":"Guizilini","year":"2020"},{"key":"10.1016\/j.neunet.2026.109049_bib0013","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9233","article-title":"Towards zero-shot scale-aware monocular depth estimation","author":"Guizilini","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0014","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.108089","article-title":"A vision-language model for multitask classification of memes","volume":"194","author":"Hossain","year":"2026","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109049_bib0015","doi-asserted-by":"crossref","unstructured":"Hu, M., Yin, W., Zhang, C., Cai, Z., Long, X., Chen, H., Wang, K., Yu, G., Shen, C., & Shen, S. (2024a). Metric3D v2: A versatile monocular geometric foundation model for zero-shot metric depth and surface normal estimation. In arxiv: 2404.15506.","DOI":"10.1109\/TPAMI.2024.3444912"},{"key":"10.1016\/j.neunet.2026.109049_bib0016","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"5594","article-title":"Learning to adapt clip for few-shot monocular depth estimation","author":"Hu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0017","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"21741","article-title":"DDP: Diffusion model for dense visual prediction","author":"Ji","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0018","series-title":"European conference on computer vision","first-page":"709","article-title":"Visual prompt tuning","author":"Jia","year":"2022"},{"key":"10.1016\/j.neunet.2026.109049_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9492","article-title":"Repurposing diffusion-based image generators for monocular depth estimation","author":"Ke","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0020","doi-asserted-by":"crossref","first-page":"479","DOI":"10.1016\/j.neunet.2021.07.007","article-title":"An efficient encoder-decoder model for portrait depth estimation from single images trained on pixel-accurate synthetic data","volume":"142","author":"Khan","year":"2021","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109049_bib0021","unstructured":"Kim, D., & Lee, S. (2024). Clip can understand depth. In arxiv: 2402.03251."},{"key":"10.1016\/j.neunet.2026.109049_bib0022","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13883","article-title":"Text-image alignment for diffusion-based perception","author":"Kondapaneni","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0023","unstructured":"Lavreniuk, M., Bhat, S. F., M\u00fcller, M., & Wonka, P. (2023). EVP: Enhanced visual perception using inverse multi-attentive feature refinement and regularized image-text alignment. In arxiv: 2312.08548."},{"key":"10.1016\/j.neunet.2026.109049_bib0024","unstructured":"Lee, J. H., Han, M.-K., Ko, D. W., & Suh, I. H. (2019). From big to small: Multi-scale local planar guidance for monocular depth estimation. In arxiv: 1907.10326."},{"key":"10.1016\/j.neunet.2026.109049_bib0025","series-title":"International conference on machine learning","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.neunet.2026.109049_bib0026","series-title":"IEEE transactions on image processing","article-title":"BinsFormer: Revisiting adaptive bins for monocular depth estimation","author":"Li","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0027","doi-asserted-by":"crossref","unstructured":"Lin, H., Peng, S., Chen, J., Peng, S., Sun, J., Liu, M., Bao, H., Feng, J., Zhou, X., & Kang, B. (2024). Prompting depth anything for 4k resolution accurate metric depth estimation. arxiv: 2412.14015.","DOI":"10.1109\/CVPR52734.2025.01591"},{"key":"10.1016\/j.neunet.2026.109049_bib0028","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107565","article-title":"Semantic discrete decoder based on adaptive pixel clustering for monocular depth estimation","volume":"189","author":"Liu","year":"2025","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109049_bib0029","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1016\/j.neunet.2023.12.020","article-title":"Joint estimation of pose, depth, and optical flow with a competition-cooperation transformer network","volume":"171","author":"Liu","year":"2024","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109049_bib0030","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4460","article-title":"Occupancy networks: Learning 3D reconstruction in function space","author":"Mescheder","year":"2019"},{"key":"10.1016\/j.neunet.2026.109049_bib0031","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"19900","article-title":"All in tokens: Unifying output space of visual tasks via soft token","author":"Ning","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0032","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10106","article-title":"UniDepth: Universal monocular metric depth estimation","author":"Piccinelli","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0033","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"283","article-title":"GeoNet: Geometric neural network for joint depth and surface normal estimation","author":"Qi","year":"2018"},{"key":"10.1016\/j.neunet.2026.109049_bib0034","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neunet.2026.109049_bib0035","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"12179","article-title":"Vision transformers for dense prediction","author":"Ranftl","year":"2021"},{"key":"10.1016\/j.neunet.2026.109049_bib0036","series-title":"Ieee transactions on pattern analysis and machine intelligence","first-page":"1623","article-title":"Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer","volume":"vol. 44","author":"Ranftl","year":"2020"},{"key":"10.1016\/j.neunet.2026.109049_bib0037","unstructured":"Reiner, D. B., Wofk, M., & M\u00fcller, M. (2023). MiDaS v3. 1\u2013a model zoo for robust monocular relative depth estimation. In arxiv: 2307.14460."},{"key":"10.1016\/j.neunet.2026.109049_bib0038","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15804","article-title":"MGNet: Monocular geometric scene understanding for autonomous driving","author":"Sch\u00f6n","year":"2021"},{"key":"10.1016\/j.neunet.2026.109049_bib0039","series-title":"IEEE transactions on multimedia","article-title":"URCDC-depth: Uncertainty rectified cross-distillation with cutflip for monocular depth estimation","author":"Shao","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0040","series-title":"Computer vision\u2013ECCV 2012: 12th european conference on computer vision, florence, italy, october 7\u201313, 2012, proceedings, part v 12","first-page":"746","article-title":"Indoor segmentation and support inference from rgbd images","author":"Silberman","year":"2012"},{"key":"10.1016\/j.neunet.2026.109049_bib0041","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"567","article-title":"Sun RGB-D: A RGB-D scene understanding benchmark suite","author":"Song","year":"2015"},{"key":"10.1016\/j.neunet.2026.109049_bib0042","doi-asserted-by":"crossref","unstructured":"Song, Z., Wang, Z., Li, B., Zhang, H., Zhu, R., Liu, L., Jiang, P.-T., & Zhang, T. (2025). DepthMaster: Taming diffusion models for monocular depth estimation. In arxiv: 2501.02576.","DOI":"10.1109\/TCSVT.2026.3681436"},{"issue":"12","key":"10.1016\/j.neunet.2026.109049_bib0043","doi-asserted-by":"crossref","first-page":"19729","DOI":"10.1109\/TITS.2024.3464528","article-title":"Weakly-supervised pavement surface crack segmentation based on dual separation and domain generalization","volume":"25","author":"Tao","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"10.1016\/j.neunet.2026.109049_bib0044","series-title":"2017 international conference on 3D vision (3DV)","first-page":"11","article-title":"Sparsity invariant cnns","author":"Uhrig","year":"2017"},{"key":"10.1016\/j.neunet.2026.109049_bib0045","doi-asserted-by":"crossref","unstructured":"Viola, M., Qu, K., Metzger, N., Ke, B., Becker, A., Schindler, K., & Obukhov, A. (2024). Marigold-DC: Zero-shot monocular depth completion with guided diffusion. In arxiv: 2412.13389.","DOI":"10.1109\/ICCV51701.2025.00509"},{"key":"10.1016\/j.neunet.2026.109049_bib0046","series-title":"2023 IEEE international conference on robotics and automation (ICRA)","first-page":"6095","article-title":"Monocular visual-inertial depth estimation","author":"Wofk","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10371","article-title":"Depth anything: Unleashing the power of large-scale unlabeled data","author":"Yang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0048","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Zhao, Z., Xu, X., Feng, J., & Zhao, H. (2024b). Depth anything v2. In arxiv: 2406.09414.","DOI":"10.52202\/079017-0688"},{"key":"10.1016\/j.neunet.2026.109049_bib0049","unstructured":"Yin, W., Wang, X., Shen, C., Liu, Y., Tian, Z., Xu, S., Sun, C., & Renyin, D. (2020). DiverseDepth: Affine-invariant depth prediction using diverse data. In arxiv: 2002.00569."},{"key":"10.1016\/j.neunet.2026.109049_bib0050","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9043","article-title":"Metric3D: Towards zero-shot metric 3D prediction from a single image","author":"Yin","year":"2023"},{"key":"10.1016\/j.neunet.2026.109049_bib0051","unstructured":"Zeng, Z., Ni, J., Wang, D., Rim, P., Chung, Y., Yang, F., Hong, B.-W., & Wong, A. (2024a). Iris: Integrating language into diffusion-based monocular depth estimation. arXiv: 2411.16750."},{"key":"10.1016\/j.neunet.2026.109049_bib0052","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9708","article-title":"WorDepth: Variational language prior for monocular depth estimation","author":"Zeng","year":"2024"},{"key":"10.1016\/j.neunet.2026.109049_bib0053","doi-asserted-by":"crossref","unstructured":"Zeng, Z., Wu, Y., Park, H., Wang, D., Yang, F., Soatto, S., Lao, D., Hong, B.-W., & Wong, A. (2024c). RSA: Resolving scale ambiguities in monocular depth estimators through language descriptions. In arxiv: 2410.02924.","DOI":"10.52202\/079017-3580"},{"key":"10.1016\/j.neunet.2026.109049_bib0054","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"6868","article-title":"Can language understand depth?","author":"Zhang","year":"2022"},{"key":"10.1016\/j.neunet.2026.109049_bib0055","doi-asserted-by":"crossref","unstructured":"Zhang, X., Ke, B., Riemenschneider, H., Metzger, N., Obukhov, A., Gross, M., Schindler, K., & Schroers, C. (2024). BetterDepth: Plug-and-play diffusion refiner for zero-shot monocular depth estimation. In arxiv: 2407.17952.","DOI":"10.52202\/079017-3451"},{"key":"10.1016\/j.neunet.2026.109049_bib0056","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"5729","article-title":"Unleashing text-to-image diffusion models for visual perception","author":"Zhao","year":"2023"},{"issue":"16","key":"10.1016\/j.neunet.2026.109049_bib0057","doi-asserted-by":"crossref","first-page":"16281","DOI":"10.1007\/s11042-024-19691-x","article-title":"An end-to-end repair-based joint training framework for weakly supervised pavement crack segmentation","volume":"84","author":"Zhou","year":"2025","journal-title":"Multimedia Tools and Applications"},{"key":"10.1016\/j.neunet.2026.109049_bib0058","doi-asserted-by":"crossref","first-page":"502","DOI":"10.1016\/j.neunet.2023.03.012","article-title":"Miper-MVS: Multi-scale iterative probability estimation with refinement for efficient multi-view stereo","volume":"162","author":"Zhou","year":"2023","journal-title":"Neural Networks"},{"key":"10.1016\/j.neunet.2026.109049_bib0059","unstructured":"Zhu, R., Wang, C., Song, Z., Liu, L., Zhang, T., & Zhang, Y. (2024). ScaleDepth: Decomposing metric depth estimation into scale prediction and relative depth estimation. In arxiv: 2407.08187."}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026005095?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026005095?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T11:20:00Z","timestamp":1784200800000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026005095"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":59,"alternative-id":["S0893608026005095"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109049","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"VGLD: Visually-guided language disambiguation for monocular depth scale recovery","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109049","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109049"}}