{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:22:04Z","timestamp":1777656124465,"version":"3.51.4"},"publisher-location":"Cham","reference-count":83,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915680","type":"print"},{"value":"9783031915697","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91569-7_14","type":"book-chapter","created":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T12:50:19Z","timestamp":1748091019000},"page":"206-225","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["EVP: Enhanced Visual Perception Using Inverse Multi-attentive Feature Refinement and\u00a0Regularized Image-Text Alignment"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2183-8833","authenticated-orcid":false,"given":"Mykola","family":"Lavreniuk","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7668-4424","authenticated-orcid":false,"given":"Shariq Farooq","family":"Bhat","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5249-8734","authenticated-orcid":false,"given":"Matthias","family":"M\u00fcller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0627-9746","authenticated-orcid":false,"given":"Peter","family":"Wonka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"14_CR1","doi-asserted-by":"publisher","unstructured":"Adams, W.J., Elder, J.H., Graf, E.W., Leyland, J., Lugtigheid, A.J., Muryy, A.: The Southampton-York natural scenes (SYNS) dataset: statistics of surface attitude. Sci. Rep. 6(1) (2016). https:\/\/doi.org\/10.1038\/srep35805","DOI":"10.1038\/srep35805"},{"key":"14_CR2","doi-asserted-by":"crossref","unstructured":"Agarwal, A., Arora, C.: Attention attention everywhere: monocular depth prediction with skip attention. arXiv preprint arXiv:2210.09071 (2022)","DOI":"10.1109\/WACV56688.2023.00581"},{"key":"14_CR3","unstructured":"Balaji, Y., et al.: eDiff-i: text-to-image diffusion models with ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)"},{"key":"14_CR4","unstructured":"Bhat, S.F., Alhashim, I., Wonka, P.: Adabins: depth estimation using adaptive bins. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4009\u20134018 (2021)"},{"key":"14_CR5","doi-asserted-by":"crossref","unstructured":"Bhat, S.F., Alhashim, I., Wonka, P.: Localbins: improving depth estimation by learning local distributions. In: European Conference on Computer Vision, pp. 480\u2013496. Springer (2022)","DOI":"10.1007\/978-3-031-19769-7_28"},{"key":"14_CR6","unstructured":"Bhat, S.F., Birkl, R., Wofk, D., Wonka, P., Muller, M.: Zoedepth: zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)"},{"key":"14_CR7","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., Efros, A.A.: Instructpix2pix: learning to follow image editing instructions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18392\u201318402 (2023)","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"14_CR8","doi-asserted-by":"crossref","unstructured":"Chen, D.J., Jia, S., Lo, Y.C., Chen, H.T., Liu, T.L.: See-through-text grouping for referring image segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00755"},{"key":"14_CR9","doi-asserted-by":"crossref","unstructured":"Costanzino, A., Ramirez, P.Z., Poggi, M., Tosi, F., Mattoccia, S., Di\u00a0Stefano, L.: Learning depth estimation for transparent and mirror surfaces. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9244\u20139255 (2023)","DOI":"10.1109\/ICCV51070.2023.00848"},{"key":"14_CR10","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat GANs on image synthesis. Adv. Neural. Inf. Process. Syst. 34, 8780\u20138794 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"14_CR11","doi-asserted-by":"crossref","unstructured":"Ding, H., Liu, C., Wang, S., Jiang, X.: VLT: vision-language transformer and query generation for referring segmentation. IEEE Trans. Pattern Anal. Mach. Intell. (TPAMI) (2022)","DOI":"10.1109\/TPAMI.2022.3217852"},{"key":"14_CR12","unstructured":"Eigen, D., Puhrsch, C., Fergus, R.: Depth map prediction from a single image using a multi-scale deep network. In: NIPS (2014)"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12873\u201312883 (2021)","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"14_CR14","doi-asserted-by":"crossref","unstructured":"Feng, G., Hu, Z., Zhang, L., Lu, H.: Encoder fusion network with co-attention embedding for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15506\u201315515 (2021)","DOI":"10.1109\/CVPR46437.2021.01525"},{"key":"14_CR15","unstructured":"Gao, P., et al.: Clip-adapter: better vision-language models with feature adapters. arXiv preprint arXiv:2110.04544 (2021)"},{"key":"14_CR16","unstructured":"Gubins, I., Veltkamp, R.: Deeply cascaded u-net for multi-task image processing. arXiv preprint arXiv:2005.00225 (2020)"},{"key":"14_CR17","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-Or, D.: Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626 (2022)"},{"key":"14_CR18","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"14_CR19","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Hu, Z., Feng, G., Sun, J., Zhang, L., Lu, H.: Bi-directional relationship inferring network for referring image segmentation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00448"},{"key":"14_CR21","unstructured":"Huang, H., Feng, Y., Shi, C., Xu, L., Yu, J., Yang, S.: Free-bloom: zero-shot text-to-video generator with LLM director and LDM animator. arXiv preprint arXiv:2309.14494 (2023)"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Huang, S., et al.: Referring image segmentation via cross-modal progressive comprehension. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01050"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Khachatryan, L., et al.: Text2video-zero: text-to-image diffusion models are zero-shot video generators. arXiv preprint arXiv:2303.13439 (2023)","DOI":"10.1109\/ICCV51070.2023.01462"},{"key":"14_CR24","unstructured":"Khan, M.O., Liang, J., Wang, C.K., Yang, S., Lou, Y.: Mesa: masked, geometric, and supervised pre-training for monocular depth estimation (2023)"},{"key":"14_CR25","doi-asserted-by":"crossref","unstructured":"Kim, N., Kim, D., Lan, C., Zeng, W., Kwak, S.: Restr: convolution-free referring image segmentation using transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18145\u201318154 (2022)","DOI":"10.1109\/CVPR52688.2022.01761"},{"key":"14_CR26","doi-asserted-by":"crossref","unstructured":"Kim, Y., Jung, H., Min, D., Sohn, K.: Deep monocular depth estimation via integration of global and local predictions, pp. 4131\u20134144 (2018)","DOI":"10.1109\/TIP.2018.2836318"},{"key":"14_CR27","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Girshick, R., He, K., Dollar, P.: Panoptic feature pyramid networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6399\u20136408 (2019)","DOI":"10.1109\/CVPR.2019.00656"},{"key":"14_CR28","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"14_CR29","doi-asserted-by":"crossref","unstructured":"Kondapaneni, N., Marks, M., Knott, M., Guimar\u00e3es, R., Perona, P.: Text-image alignment for diffusion-based perception (2023)","DOI":"10.1109\/CVPR52733.2024.01317"},{"key":"14_CR30","unstructured":"Lavreniuk, M.: Spidepth: strengthened pose information for self-supervised monocular depth estimation. arXiv preprint arXiv:2404.12501 (2024)"},{"key":"14_CR31","unstructured":"Lee, J.H., Han, M.K., Ko, D.W., Suh, I.H.: From big to small: multi-scale local planar guidance for monocular depth estimation. arXiv preprint arXiv:1907.10326 (2019)"},{"key":"14_CR32","doi-asserted-by":"crossref","unstructured":"Lee, J., et al.: Slabins: fisheye depth estimation using slanted bins on road environments. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8765\u20138774 (2023)","DOI":"10.1109\/ICCV51070.2023.00805"},{"key":"14_CR33","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv preprint arXiv:2301.12597 (2023)"},{"key":"14_CR34","doi-asserted-by":"crossref","unstructured":"Li, X., Wang, W., Hu, X., Yang, J.: Selective kernel networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00060"},{"key":"14_CR35","unstructured":"Li, Z., Wang, X., Liu, X., Jiang, J.: Binsformer: revisiting adaptive bins for monocular depth estimation. arXiv preprint arXiv:2204.00987 (2022)"},{"key":"14_CR36","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Dollar, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"14_CR37","doi-asserted-by":"crossref","unstructured":"Liu, C., Ding, H., Jian, X.: Gres: generalized referring expression segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23592\u201323601 (2023)","DOI":"10.1109\/CVPR52729.2023.02259"},{"key":"14_CR38","doi-asserted-by":"crossref","unstructured":"Liu, H., Shen, X., Shang, F., Ge, F., Wang, F.: Cu-net: cascaded u-net with loss weighted sampling for brain tumor segmentation. In: Multimodal Brain Image Analysis and Mathematical Foundations of Computational Anatomy, pp. 102\u2013111 (2019)","DOI":"10.1007\/978-3-030-33226-6_12"},{"key":"14_CR39","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: Polyformer: referring image segmentation as sequential polygon generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18653\u201318663 (2023)","DOI":"10.1109\/CVPR52729.2023.01789"},{"key":"14_CR40","doi-asserted-by":"crossref","unstructured":"Liu, Z., et\u00a0al.: Swin transformer v2: scaling up capacity and resolution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12009\u201312019 (2022)","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"14_CR41","doi-asserted-by":"crossref","unstructured":"Lugmayr, A., Danelljan, M., Romero, A., Yu, F., Timofte, R., Van\u00a0Gool, L.: Repaint: inpainting using denoising diffusion probabilistic models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11461\u201311471 (2022)","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"14_CR42","doi-asserted-by":"crossref","unstructured":"Luo, G., et al.: Multi-task collaborative network for joint referring expression comprehension and segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10034\u201310043 (2020)","DOI":"10.1109\/CVPR42600.2020.01005"},{"key":"14_CR43","doi-asserted-by":"crossref","unstructured":"Mei, H., et al.: Don\u2019t hit me! glass detection in real-world scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.00374"},{"key":"14_CR44","doi-asserted-by":"crossref","unstructured":"Menze, M., Geiger, A.: Object scene flow for autonomous vehicles. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2015)","DOI":"10.1109\/CVPR.2015.7298925"},{"key":"14_CR45","unstructured":"Nichol, A., et al.: Glide: towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741 (2021)"},{"key":"14_CR46","unstructured":"Nichol, A.Q., Dhariwal, P.: Improved denoising diffusion probabilistic models. In: International Conference on Machine Learning, pp. 8162\u20138171. PMLR (2021)"},{"key":"14_CR47","doi-asserted-by":"crossref","unstructured":"Ning, J., et al.: All in tokens: unifying output space of visual tasks via soft token. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 19900\u201319910 (2023)","DOI":"10.1109\/ICCV51070.2023.01822"},{"key":"14_CR48","doi-asserted-by":"crossref","unstructured":"Patil, V., Sakaridis, C., Liniger, A., Gool, L.V.: P3depth: monocular depth estimation with a piecewise planarity prior. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1610\u20131621 (2022)","DOI":"10.1109\/CVPR52688.2022.00166"},{"key":"14_CR49","doi-asserted-by":"crossref","unstructured":"Piccinelli, L., Sakaridis, C., Yu, F.: iDisc: internal discretization for monocular depth estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21477\u201321487 (2023)","DOI":"10.1109\/CVPR52729.2023.02057"},{"key":"14_CR50","unstructured":"Podell, D., et al.: Sdxl: improving latent diffusion models for high-resolution image synthesis (2023)"},{"key":"14_CR51","unstructured":"Poole, B., Jain, A., Barron, J.T., Mildenhall, B.: Dreamfusion: text-to-3D using 2D diffusion. arXiv preprint arXiv:2209.14988 (2022)"},{"key":"14_CR52","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763 (2021)"},{"key":"14_CR53","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"14_CR54","doi-asserted-by":"crossref","unstructured":"Ranftl, R., Bochkovskiy, A., Koltun, V.: Vision transformers for dense prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 12179\u201312188 (2021)","DOI":"10.1109\/ICCV48922.2021.01196"},{"key":"14_CR55","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"14_CR56","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015, pp. 234\u2013241. Springer, Cham (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"14_CR57","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., et al.: Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural. Inf. Process. Syst. 35, 36479\u201336494 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"14_CR58","unstructured":"Sarwari, K., Laine, F., Tomlin, C.: Progress and Proposals: A Case Study of Monocular Depth Estimation. Master\u2019s thesis, EECS Department, University of California, Berkeley (2021). http:\/\/www2.eecs.berkeley.edu\/Pubs\/TechRpts\/2021\/EECS-2021-32.html"},{"key":"14_CR59","doi-asserted-by":"crossref","unstructured":"Shao, S., Pei, Z., Chen, W., Wu, X., Li, Z.: Nddepth: normal-distance assisted monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7931\u20137940 (2023)","DOI":"10.1109\/ICCV51070.2023.00729"},{"key":"14_CR60","doi-asserted-by":"crossref","unstructured":"Shao, S., Pei, Z., Wu, X., Liu, Z., Chen, W., Li, Z.: Iebins: iterative elastic bins for monocular depth estimation. arXiv preprint arXiv:2309.14137 (2023)","DOI":"10.1007\/s11263-024-02293-3"},{"key":"14_CR61","doi-asserted-by":"crossref","unstructured":"Silberman, N., Hoiem, D., Kohli, P., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: Computer Vision \u2013 ECCV 2012, pp. 746\u2013760. Springer, Heidelberg (2012)","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"14_CR62","unstructured":"Spencer, J., Russell, C., Hadfield, S., Bowden, R.: Deconstructing self-supervised monocular reconstruction: the design decisions that matter. Trans. Mach. Learn. Res. (2022)"},{"key":"14_CR63","unstructured":"Spencer, J., et al.: The third monocular depth estimation challenge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops, pp. 1\u201314 (2024)"},{"key":"14_CR64","unstructured":"Vasiljevic, I., et al.: DIODE: A Dense Indoor and Outdoor DEpth Dataset. CoRR abs\/1908.00463 (2019). http:\/\/arxiv.org\/abs\/1908.00463"},{"key":"14_CR65","doi-asserted-by":"crossref","unstructured":"Wang, Q., Wu, B., Zhu, P., Li, P., Zuo, W., Hu, Q.: ECA-net: efficient channel attention for deep convolutional neural networks. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"14_CR66","doi-asserted-by":"crossref","unstructured":"Wang, Y., Liang, Y., Xu, H., Jiao, S., Yu, H.: Sqldepth: generalizable self-supervised fine-structured monocular depth estimation (2023)","DOI":"10.1609\/aaai.v38i6.28383"},{"key":"14_CR67","doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: CRIS: clip-driven referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11686\u201311695 (2022)","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"14_CR68","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.Y., Kweon, I.S.: CBAM: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"14_CR69","doi-asserted-by":"crossref","unstructured":"Xie, Z., Geng, Z., Hu, J., Zhang, Z., Hu, H., Cao, Y.: Revealing the dark secrets of masked image modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14475\u201314485 (2023)","DOI":"10.1109\/CVPR52729.2023.01391"},{"key":"14_CR70","doi-asserted-by":"crossref","unstructured":"Yan, B., et al.: Universal instance perception as object discovery and retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15325\u201315336 (2023)","DOI":"10.1109\/CVPR52729.2023.01471"},{"key":"14_CR71","doi-asserted-by":"crossref","unstructured":"Yang, L., Kang, B., Huang, Z., Xu, X., Feng, J., Zhao, H.: Depth anything: unleashing the power of large-scale unlabeled data. arXiv preprint arXiv:2401.10891 (2024)","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"14_CR72","doi-asserted-by":"crossref","unstructured":"Yang, X., Ma, Z., Ji, Z., Ren, Z.: Gedepth: ground embedding for monocular depth estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 12719\u201312727 (2023)","DOI":"10.1109\/ICCV51070.2023.01168"},{"key":"14_CR73","doi-asserted-by":"crossref","unstructured":"Yang, X., et al.: Polymax: general dense prediction with mask transformer (2023)","DOI":"10.1109\/WACV57701.2024.00109"},{"key":"14_CR74","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, J., Tang, Y., Chen, K., Zhao, H., Torr, P.H.: Lavt: languageaware vision transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18155\u201318165 (2022)","DOI":"10.1109\/CVPR52688.2022.01762"},{"key":"14_CR75","doi-asserted-by":"crossref","unstructured":"Yu, L., et al.: Mattnet: modular attention network for referring expression comprehension. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11307\u201311315 (2018)","DOI":"10.1109\/CVPR.2018.00142"},{"key":"14_CR76","doi-asserted-by":"crossref","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 69\u201385 (2016)","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"14_CR77","doi-asserted-by":"crossref","unstructured":"Yuan, W., Gu, X., Dai, Z., Zhu, S., Tan, P.: New CRFs: neural window fully-connected CRFs for monocular depth estimation. arXiv preprint arXiv:2203.01502 (2022)","DOI":"10.1109\/CVPR52688.2022.00389"},{"key":"14_CR78","unstructured":"Zama\u00a0Ramirez, P., et al.: Tricky 2024 challenge on monocular depth from images of specular and transparent surfaces. In: European Conference on Computer Vision Workshops ECCVW (2024)"},{"issue":"1","key":"14_CR79","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1109\/TPAMI.2023.3323858","volume":"46","author":"P Zama Ramirez","year":"2024","unstructured":"Zama Ramirez, P., et al.: Booster: a benchmark for depth from images of specular and transparent surfaces. IEEE Trans. Pattern Anal. Mach. Intell. 46(1), 85\u2013102 (2024). https:\/\/doi.org\/10.1109\/TPAMI.2023.3323858","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"14_CR80","doi-asserted-by":"crossref","unstructured":"Zama\u00a0Ramirez, P., Tosi, F., Poggi, M., Salti, S., Mattoccia, S., Di\u00a0Stefano, L.: Open challenges in deep stereo: the booster dataset. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 21168\u201321178 (2022)","DOI":"10.1109\/CVPR52688.2022.02049"},{"key":"14_CR81","doi-asserted-by":"crossref","unstructured":"Zhao, W., Rao, Y., Liu, Z., Liu, B., Zhou, J., Lu, J.: Unleashing text-to-image diffusion models for visual perception. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5729\u20135739 (2023)","DOI":"10.1109\/ICCV51070.2023.00527"},{"key":"14_CR82","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Siddiquee, M.M.R., Tajbakhsh, N., Liang, J.: Unet++: a nested u-net architecture for medical image segmentation. In: Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support, pp. 3\u201311 (2018)","DOI":"10.1007\/978-3-030-00889-5_1"},{"key":"14_CR83","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Siddiquee, M.M.R., Tajbakhsh, N., Liang, J.: Unet++: redesigning skip connections to exploit multiscale features in image segmentation. IEEE Trans. Med. Imaging 1856\u20131867 (2019)","DOI":"10.1109\/TMI.2019.2959609"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91569-7_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T12:50:33Z","timestamp":1748091033000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91569-7_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915680","9783031915697"],"references-count":83,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91569-7_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}