{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T17:03:23Z","timestamp":1784567003465,"version":"3.55.0"},"reference-count":89,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T00:00:00Z","timestamp":1779321600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T00:00:00Z","timestamp":1779321600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001691","name":"Japan Society for the Promotion of Science","doi-asserted-by":"publisher","award":["23K28164"],"award-info":[{"award-number":["23K28164"]}],"id":[{"id":"10.13039\/501100001691","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001695","name":"Japan Science and Technology Corporation","doi-asserted-by":"publisher","award":["JPMJCR22D1"],"award-info":[{"award-number":["JPMJCR22D1"]}],"id":[{"id":"10.13039\/501100001695","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11263-026-02813-3","type":"journal-article","created":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T03:52:41Z","timestamp":1779335561000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Training-Free Open-Vocabulary Semantic Segmentation with Context Pyramid Refinement"],"prefix":"10.1007","volume":"134","author":[{"given":"Jialei","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2644-4457","authenticated-orcid":false,"given":"Qi","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenzhen","family":"Quan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongyue","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xu","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hiroshi","family":"Murase","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daisuke","family":"Deguchi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,21]]},"reference":[{"key":"2813_CR1","doi-asserted-by":"crossref","unstructured":"Achanta, R., Shaji, A., Smith, K., Lucchi, A., Fua, P., & S\u00fcsstrunk, S. (2012). Slic superpixels compared to state-of-the-art superpixel methods. TPAMI.","DOI":"10.1109\/TPAMI.2012.120"},{"key":"2813_CR2","doi-asserted-by":"crossref","unstructured":"Bi, H., Feng, Y., Mao, Y., Pei, J., Diao, W., Wang, H., & Sun, X. (2024). Agmtr: Agent mining transformer for few-shot segmentation in remote sensing IJCV.","DOI":"10.1007\/s11263-024-02252-y"},{"key":"2813_CR3","doi-asserted-by":"crossref","unstructured":"Bousselham, W., Petersen, F., Ferrari, V., & Kuehne, H. (2024). Grounding everything: Emerging localization properties in vision-language transformers. CVPR.","DOI":"10.1109\/CVPR52733.2024.00367"},{"key":"2813_CR4","doi-asserted-by":"crossref","unstructured":"Caesar, H., Uijlings, J., & Ferrari, V. (2018). Coco-stuff: Thing and stuff classes in context. CVPR.","DOI":"10.1109\/CVPR.2018.00132"},{"key":"2813_CR5","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., J., & A. (2021). Emerging properties in self-supervised vision transformers. ICCV.","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"2813_CR6","doi-asserted-by":"crossref","unstructured":"Cha, J., Mun, J., & Roh, B. (2023). Learning to generate text-grounded mask for open-world semantic segmentation from only image-text pairs. CVPR.","DOI":"10.1109\/CVPR52729.2023.01074"},{"key":"2813_CR7","doi-asserted-by":"crossref","unstructured":"Chen, J., Quan, Z., Zhang, C., Zheng, Xu., Deguchi, D., & Murase, H. (2025). CLIP-to-Seg Distillation for Zero-shot Semantic Segmentation. IEEE Transactions on Circuits and Systems for Video Technology. IEEE","DOI":"10.1109\/TCSVT.2025.3616588"},{"issue":"3","key":"2813_CR8","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1007\/s11263-025-02648-4","volume":"134","author":"J Chen","year":"2026","unstructured":"Chen, J., Deguchi, D., Li, D., Zheng, X., Ito, S., Murase, H., & Fan, Q. (2026). Semantic-centric alignment for zero-shot panoptic segmentation with limited data. International Journal of Computer Vision, 134(3), 122.","journal-title":"International Journal of Computer Vision"},{"key":"2813_CR9","unstructured":"Chen, J., Zheng, X., Li, D., Yi, C., Ito, S., Pani Paudel, D., Gool, L.V., Murase, H., & Deguchi, D. (2025a). Split matching for inductive zero-shot semantic segmentation. In 36th British Machine Vision Conference 2025, Sheffield, UK, November 24-27, 2025. BMVA"},{"key":"2813_CR10","doi-asserted-by":"crossref","unstructured":"Chen, J., Zheng, Xu., Paudel, D. P., Gool, L.V., Murase, H., & Deguchi, D. (2025b). BiXFormer: a robust framework for maximizing modality effectiveness in multi-modal semantic segmentation. arXiv preprint arXiv:2506.03675.","DOI":"10.1109\/TMM.2026.3703333"},{"key":"2813_CR11","doi-asserted-by":"crossref","unstructured":"Chen, J., Deguchi, D., Zhang, C., Zheng, X., & Murase, H. (2024). Frozen is better than learning: A new design of prototype-based classifier for semantic segmentation. PR.","DOI":"10.2139\/ssrn.4617170"},{"key":"2813_CR12","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020). A simple framework for contrastive learning of visual representations. ICML."},{"key":"2813_CR13","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Papandreou, G., Kokkinos, I., Murphy, K., Y., & A. L. (2017). Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. TPAMI.","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"2813_CR14","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., A., & H. (2018). Encoder-decoder with atrous separable convolution for semantic image segmentation. ECCV.","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"2813_CR15","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A. G., Kirillov, A., & Girdhar, R. (2022). Masked-attention mask transformer for universal image segmentation. CVPR.","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"2813_CR16","unstructured":"Cheng, B., Schwing, A., & Kirillov, A. (2021). Per-pixel classification is not all you need for semantic segmentation. NeurIPS."},{"key":"2813_CR17","doi-asserted-by":"crossref","unstructured":"Cherti, M., Beaumont, R., Wightman, R., Wortsman, M., Ilharco, G., Gordon, C., Schuhmann, C., Schmidt, L., & Jitsev, J. (2023). Reproducible scaling laws for contrastive language-image learning. CVPR.","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"2813_CR18","doi-asserted-by":"crossref","unstructured":"Cho, S., Shin, H., Hong, S., An, S., Lee, S., Arnab, A., Seo, P. H., & Kim, S. (2023). Cat-seg: Cost aggregation for open-vocabulary semantic segmentation arXiv preprint.","DOI":"10.1109\/CVPR52733.2024.00394"},{"key":"2813_CR19","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., & Schiele, B. (2016). The cityscapes dataset for semantic urban scene understanding. CVPR.","DOI":"10.1109\/CVPR.2016.350"},{"key":"2813_CR20","doi-asserted-by":"crossref","unstructured":"Dao, S. D., Shi, H., Phung, D. Q., & Cai, J. (2024). Ca-ovs: Cluster and adapt mask proposals for open-vocabulary semantic segmentation. MmAsia.","DOI":"10.1145\/3696409.3700213"},{"key":"2813_CR21","doi-asserted-by":"crossref","unstructured":"Ding, J., Xue, N., Xia, G.-S., & Dai, D. (2022). Decoupling zero-shot semantic segmentation. CVPR.","DOI":"10.1109\/CVPR52688.2022.01129"},{"key":"2813_CR22","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., & Gelly, S. (2020). An image is worth 16x16 words: Transformers for image recognition at scale arXiv preprint."},{"key":"2813_CR23","doi-asserted-by":"crossref","unstructured":"Everingham, M., Eslami, S. M. A., Van Gool, L., Williams, C. K. I., Winn, J., & Zisserman, A. (2015). The pascal visual object classes challenge: A retrospective IJCV.","DOI":"10.1007\/s11263-014-0733-5"},{"key":"2813_CR24","doi-asserted-by":"crossref","unstructured":"Ghiasi, G., Gu, X., Cui, Y., & Lin, T.-Y. (2022). Scaling open-vocabulary image segmentation with image-level labels. ECCV.","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"2813_CR25","doi-asserted-by":"crossref","unstructured":"Guo, M.-H., Lu, C.-Z., Hou, Q., Liu, Z., Cheng, M.-M., & Hu, S.-M. (2022). Segnext: Rethinking convolutional attention design for semantic segmentation. NeurIPS.","DOI":"10.52202\/068431-0084"},{"key":"2813_CR26","doi-asserted-by":"crossref","unstructured":"Hajimiri, S., Ben Ayed, I., & Dolz, J. (2025). Pay attention to your neighbours: Training-free open-vocabulary semantic segmentation. WACV.","DOI":"10.1109\/WACV61041.2025.00495"},{"key":"2813_CR27","doi-asserted-by":"crossref","unstructured":"Han, K., Liu, Y., Liew, J.H., Ding, H., Liu, J., Wang, Y., Tang, Y., Yang, Y., Feng, J., Zhao, Y., et al. (2023). Global knowledge calibration for fast open-vocabulary segmentation. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.00080"},{"key":"2813_CR28","doi-asserted-by":"crossref","unstructured":"Han, C., Zhong, Y., Li, D., Han, K., & Ma, L. (2023). Zero-shot semantic segmentation with decoupled one-pass network. arXiv preprint arXiv:2304.01198.","DOI":"10.1109\/ICCV51070.2023.00106"},{"key":"2813_CR29","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., & Girshick, R. (2020). Momentum contrast for unsupervised visual representation learning. CVPR.","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"2813_CR30","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. CVPR.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2813_CR31","unstructured":"Hu, X., Saining, X., Xiaoqing, E.T., Po-Yao, H., Russell, H., Vasu, S., Shang-Wen, L., Gargi, G., Luke, Z., & Christoph, F. (2023). Demystifying clip data. arXiv preprint arXiv:2309.16671."},{"key":"2813_CR32","doi-asserted-by":"crossref","unstructured":"Jatavallabhula, K.M., Kuwajerwala, A., Gu, Q., Omama, M., Chen, T., Maalouf, A., Li, S., Iyer, G., Saryazdi, S., Keetha, N., & others (2023). Conceptfusion: Open-set multimodal 3d mapping. arXiv preprint arXiv:2302.07241.","DOI":"10.15607\/RSS.2023.XIX.066"},{"key":"2813_CR33","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.-T., Parekh, Z., Pham, H., Le, Q., Sung, Y.-H., Li, Z., & Duerig, T. (2021). Scaling up visual and vision-language representation learning with noisy text supervision. ICML."},{"key":"2813_CR34","doi-asserted-by":"crossref","unstructured":"Kim, C., Ju, D., Han, W., Yang, M.-H., & Hwang, S. J. (2025). Distilling spectral graph for object-context aware open-vocabulary semantic segmentation. ICCV.","DOI":"10.1109\/CVPR52734.2025.01400"},{"key":"2813_CR35","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A. C., & Lo, W.-Y. (2023). Segment anything. In: ICCV.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2813_CR36","doi-asserted-by":"crossref","unstructured":"Lan, M., Chen, C., Ke, Y., Wang, X., Feng, L., & Zhang, W. (2024). Clearclip: Decomposing clip representations for dense vision-language inference. ECCV.","DOI":"10.1007\/978-3-031-72970-6_9"},{"key":"2813_CR37","doi-asserted-by":"crossref","unstructured":"Lan, M., Chen, C., Ke, Y., Wang, X., Feng, L., & Zhang, W. (2024). Proxyclip: Proxy attention improves clip for open-vocabulary segmentation. ECCV.","DOI":"10.1007\/978-3-031-73113-6_5"},{"key":"2813_CR38","unstructured":"Li, J., Chen, P., Qian, S., & Jia, J. (2023). Tagclip: Improving discrimination ability of open-vocabulary semantic segmentation arXiv preprint."},{"key":"2813_CR39","unstructured":"Li, J., Li, D., Xiong, C., & Hoi, S. (2022). Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML."},{"key":"2813_CR40","unstructured":"Li, Y., Wang, H., Duan, Y., & Li, X. (2023). Clip surgery for better explainability with enhancement in open-vocabulary tasks. arXiv preprint arXiv:2304.05653."},{"key":"2813_CR41","doi-asserted-by":"crossref","unstructured":"Liang, F., Wu, B., Dai, X., Li, K., Zhao, Y., Zhang, H., Zhang, P., Vajda, P., & Marculescu, D. (2023). Open-vocabulary semantic segmentation with mask-adapted clip. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.00682"},{"key":"2813_CR42","doi-asserted-by":"crossref","unstructured":"Lin, Y., Chen, M., Zhang, K., Li, H., Li, M., Yang, Z., Lv, D., Lin, B., Liu, H., & Cai, D. (2024). Tagclip: A local-to-global framework to enhance open-vocabulary multi-label classification of clip without training. In: AAAI.","DOI":"10.1609\/aaai.v38i4.28139"},{"key":"2813_CR43","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., & Belongie, S. (2017). Feature pyramid networks for object detection. CVPR.","DOI":"10.1109\/CVPR.2017.106"},{"key":"2813_CR44","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., & Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: ICCV.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2813_CR45","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T., & Xie, S. (2022). A convnet for the 2020s. CVPR.","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2813_CR46","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wang, G., Zhang, J., Liu, Q., & Huang, D. (2025). Unveiling the knowledge of clip for training-free open-vocabulary semantic segmentation. In: AAAI.","DOI":"10.1609\/aaai.v39i6.32602"},{"key":"2813_CR47","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., D., & T. (2015). Fully convolutional networks for semantic segmentation. CVPR.","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"2813_CR48","unstructured":"Luo, H., Bao, J., Wu, Y., He, X., & Li, T. (2023). Segclip: Patch aggregation with learnable centers for open-vocabulary semantic segmentation. ICML."},{"key":"2813_CR49","doi-asserted-by":"crossref","unstructured":"Mottaghi, R., Chen, X., Liu, X., Cho, N.-G., Lee, S.-W., Fidler, S., Urtasun, R., Y., & A. (2014). The role of context for object detection and semantic segmentation in the wild. CVPR.","DOI":"10.1109\/CVPR.2014.119"},{"key":"2813_CR50","doi-asserted-by":"crossref","unstructured":"Mukhoti, J., Lin, T.-Y., Poursaeed, O., Wang, R., Shah, A., Torr, P. H., & Lim, S.-N. (2023). Open vocabulary semantic segmentation with patch aligned contrastive learning. CVPR.","DOI":"10.1109\/CVPR52729.2023.01860"},{"key":"2813_CR51","doi-asserted-by":"crossref","unstructured":"Nara, R., Lin, Y.-C., Nozawa, Y., Ng, Y., Itoh, G., Torii, O., & Matsui, Y. (2024). Revisiting relevance feedback for clip-based interactive image retrieval. ECCV.","DOI":"10.1007\/978-3-031-91585-7_1"},{"key":"2813_CR52","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., & others (2021). Learning transferable visual models from natural language supervision. In: ICML."},{"key":"2813_CR53","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., Huang, Z., Karpathy, A., Khosla, A., Bernstein, M., & others (2015). Imagenet large scale visual recognition challenge. IJCV.","DOI":"10.1007\/s11263-015-0816-y"},{"key":"2813_CR54","doi-asserted-by":"crossref","unstructured":"Sain, A., Bhunia, A. K., Chowdhury, P. N., Koley, S., Xiang, T., & Song, Y.-Z. (2023). Clip for all things zero-shot sketch-based image retrieval, fine-grained or not. CVPR.","DOI":"10.1109\/CVPR52729.2023.00271"},{"key":"2813_CR55","doi-asserted-by":"crossref","unstructured":"Shao, T., Tian, Z., Zhao, H., & Su, J. (2025). Explore the potential of clip for training-free open vocabulary semantic segmentation. ECCV.","DOI":"10.1007\/978-3-031-73016-0_9"},{"key":"2813_CR56","unstructured":"Shen, C., Liu, Y., & Zhai, W. (2024). ColCLIP: Enhancing fine-grained image retrieval with pre-trained embeddings. ICLR."},{"key":"2813_CR57","doi-asserted-by":"crossref","unstructured":"Shi, H., Dao, S. D., & Cai, J. (2025). Llmformer: Large language model for open-vocabulary semantic segmentation IJCV.","DOI":"10.1007\/s11263-024-02171-y"},{"key":"2813_CR58","doi-asserted-by":"crossref","unstructured":"Shin, H., Kim, C., Hong, S., Cho, S., Arnab, A., Seo, P.H., & Kim, S. (2024). Towards open-vocabulary semantic segmentation without semantic labels. NeurIPS.","DOI":"10.52202\/079017-0290"},{"key":"2813_CR59","doi-asserted-by":"crossref","unstructured":"Shin, G., Xie, W., A., & S. (2022). Reco: Retrieve and co-segment for zero-shot transfer. NeurIPS.","DOI":"10.52202\/068431-2446"},{"key":"2813_CR60","doi-asserted-by":"crossref","unstructured":"Siam, M. (2025). Temporal transductive inference for few-shot video object segmentation. IJCV.","DOI":"10.1007\/s11263-025-02390-x"},{"key":"2813_CR61","unstructured":"Sun, Q., Fang, Y., Wu, L., Wang, X., & Cao, Y. (2023). Eva-clip: Improved training techniques for clip at scale. arXiv"},{"key":"2813_CR62","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., & Rabinovich, A. (2015). Going deeper with convolutions. CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"2813_CR63","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. NeurIPS."},{"key":"2813_CR64","doi-asserted-by":"crossref","unstructured":"Wang, Z., Feng, T., Lyu, F., Shang, F., Feng, W., & Wan, L. (2025). Dual semantic guidance for open vocabulary semantic segmentation. In: CVPR.","DOI":"10.1109\/CVPR52734.2025.01882"},{"key":"2813_CR65","doi-asserted-by":"crossref","unstructured":"Wang, F., Mei, J., Y., & A. (2024). Sclip: Rethinking self-attention for dense vision-language inference. ECCV.","DOI":"10.1007\/978-3-031-72664-4_18"},{"key":"2813_CR66","doi-asserted-by":"crossref","unstructured":"Wang, H., Vasu, P. K. A., Faghri, F., Vemulapalli, R., Farajtabar, M., Mehta, S., Rastegari, M., Tuzel, O., & Pouransari, H. (2024). Sam-clip: Merging vision foundation models towards semantic and spatial understanding. CVPR.","DOI":"10.1109\/CVPRW63382.2024.00367"},{"key":"2813_CR67","unstructured":"Wu, S., Zhang, W., Xu, L., Jin, S., Li, X., Liu, W., & Loy, C.C. (2024). CLIPSelf: Vision transformer distills itself for open-vocabulary dense prediction. In: ICLR."},{"key":"2813_CR68","doi-asserted-by":"crossref","unstructured":"Wysocza\u0144ska, M., Ramamonjisoa, M., Trzci\u0144ski, T., & Sim\u00e9oni, O. (2024). Clip-diy: Clip dense inference yields open-vocabulary semantic segmentation for-free. WACV.","DOI":"10.1109\/WACV57701.2024.00143"},{"key":"2813_CR69","doi-asserted-by":"crossref","unstructured":"Wysocza\u0144ska, M., Sim\u00e9oni, O., Ramamonjisoa, M., Bursuc, A., Trzci\u0144ski, T., & P\u00e9rez, P. (2024). Clip-dinoiser: Teaching clip a few dino tricks for open-vocabulary semantic segmentation. ECCV.","DOI":"10.1007\/978-3-031-73030-6_18"},{"key":"2813_CR70","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., Li, L. E., & Huang, G. (2022). Vision transformer with deformable attention. CVPR.","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"2813_CR71","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J. M., & Luo, P. (2021). Segformer: Simple and efficient design for semantic segmentation with transformers. NeurIPS."},{"key":"2813_CR72","doi-asserted-by":"crossref","unstructured":"Xie, H., Wang, C., Zhao, J., Liu, Y., Dan, J., Fu, C., & Sun, B. (2024). Prcl: Probabilistic representation contrastive learning for semi-supervised semantic segmentation IJCV.","DOI":"10.1007\/s11263-024-02016-8"},{"key":"2813_CR73","doi-asserted-by":"crossref","unstructured":"Xing, Y., Kang, J., Xiao, A., Nie, J., Shao, L., & Lu, S. (2024). Rewrite caption semantics: Bridging semantic gaps for language-supervised semantic segmentation. NeurIPS.","DOI":"10.52202\/075280-3011"},{"key":"2813_CR74","doi-asserted-by":"crossref","unstructured":"Xu, J., De Mello, S., Liu, S., Byeon, W., Breuel, T., Kautz, J., W., & X. (2022). Groupvit: Semantic segmentation emerges from text supervision. CVPR.","DOI":"10.1109\/CVPR52688.2022.01760"},{"key":"2813_CR75","doi-asserted-by":"crossref","unstructured":"Xu, J., Hou, J., Zhang, Y., Feng, R., Wang, Y., Qiao, Y., & Xie, W. (2023). Learning open-vocabulary semantic segmentation models from natural language supervision. CVPR.","DOI":"10.1109\/CVPR52729.2023.00287"},{"key":"2813_CR76","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhang, Z., Wei, F., Hu, H., & Bai, X. (2023). Side adapter network for open-vocabulary semantic segmentation. CVPR.","DOI":"10.1109\/CVPR52729.2023.00288"},{"key":"2813_CR77","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhang, Z., Wei, F., Lin, Y., Cao, Y., Hu, H., & Bai, X. (2022). A simple baseline for open-vocabulary semantic segmentation with pre-trained vision-language model. ECCV.","DOI":"10.1007\/978-3-031-19818-2_42"},{"key":"2813_CR78","doi-asserted-by":"crossref","unstructured":"Yang, Y., Deng, J., Li, W., & Duan, L. (2025). Resclip: Residual attention for training-free dense vision-language inference. CVPR.","DOI":"10.1109\/CVPR52734.2025.02789"},{"key":"2813_CR79","doi-asserted-by":"crossref","unstructured":"Ye, C., Zhuge, Y., & Zhang, P. (2025). Towards open-vocabulary remote sensing image semantic segmentation. AAAI.","DOI":"10.1609\/aaai.v39i9.33022"},{"key":"2813_CR80","doi-asserted-by":"crossref","unstructured":"Zeng, Q., Lu, Z., Xie, Y., & Xia, Y. (2025). Pick: Predict and mask for semi-supervised medical image segmentation IJCV.","DOI":"10.1007\/s11263-024-02328-9"},{"key":"2813_CR81","doi-asserted-by":"crossref","unstructured":"Zhai, X., Mustafa, B., Kolesnikov, A., & Beyer, L. (2023). Sigmoid loss for language image pre-training. ICCV.","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"2813_CR82","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Guo, M.-H., Wang, M., & Hu, S.-M. (2024). Exploring regional clues in clip for zero-shot semantic segmentation. CVPR.","DOI":"10.1109\/CVPR52733.2024.00315"},{"key":"2813_CR83","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Huang, X., Ma, J., Li, Z., Luo, Z., Xie, Y., Qin, Y., Luo, T., Li, Y., Liu, S., et al. (2023). Recognize anything: A strong image tagging model. arXiv preprint arXiv:2306.03514.","DOI":"10.1109\/CVPRW63382.2024.00179"},{"key":"2813_CR84","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., & Jia, J. (2017). Pyramid scene parsing network. CVPR.","DOI":"10.1109\/CVPR.2017.660"},{"key":"2813_CR85","doi-asserted-by":"crossref","unstructured":"Zheng, S., Lu, J., Zhao, H., Zhu, X., Luo, Z., Wang, Y., Fu, Y., Feng, J., Xiang, T., Torr, P.H., et al. (2021). Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: CVPR.","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"2813_CR86","doi-asserted-by":"crossref","unstructured":"Zheng, X., Luo, Y., Zhou, P., W., & L. (2025). Distilling efficient vision transformers from cnns for semantic segmentation. PR.","DOI":"10.2139\/ssrn.4782766"},{"key":"2813_CR87","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Lei, Y., Zhang, B., Liu, L., & Liu, Y. (2023). Zegclip: Towards adapting clip for zero-shot semantic segmentation. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01075"},{"key":"2813_CR88","doi-asserted-by":"crossref","unstructured":"Zhou, C., Loy, C. C., & Dai, B. (2022). Extract free dense labels from clip. ECCV.","DOI":"10.1007\/978-3-031-19815-1_40"},{"key":"2813_CR89","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., & Torralba, A. (2017). Scene parsing through ade20k dataset. CVPR.","DOI":"10.1109\/CVPR.2017.544"}],"updated-by":[{"DOI":"10.1007\/s11263-026-02917-w","type":"correction","label":"Correction","source":"publisher","updated":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T00:00:00Z","timestamp":1784505600000}}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02813-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02813-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02813-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T16:17:09Z","timestamp":1784564229000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02813-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,21]]},"references-count":89,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2813"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02813-3","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,21]]},"assertion":[{"value":"10 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 June 2026","order":5,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Update","order":6,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The original PDF version of this article was revised due to data differences in multiple places in Table 1, 2 and 3","order":7,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 July 2026","order":8,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Correction","order":9,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"A Correction to this paper has been published:","order":10,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"https:\/\/doi.org\/10.1007\/s11263-026-02917-w","URL":"https:\/\/doi.org\/10.1007\/s11263-026-02917-w","order":11,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"281"}}