{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:09:01Z","timestamp":1776881341044,"version":"3.51.2"},"reference-count":84,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,3,17]],"date-time":"2025-03-17T00:00:00Z","timestamp":1742169600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,17]],"date-time":"2025-03-17T00:00:00Z","timestamp":1742169600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100010909","name":"Excellent Young Scientists Fund","doi-asserted-by":"publisher","award":["62222112"],"award-info":[{"award-number":["62222112"]}],"id":[{"id":"10.13039\/501100010909","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s11263-025-02416-4","type":"journal-article","created":{"date-parts":[[2025,3,17]],"date-time":"2025-03-17T18:23:54Z","timestamp":1742235834000},"page":"4669-4689","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Not All Pixels are Equal: Learning Pixel Hardness for Semantic Segmentation"],"prefix":"10.1007","volume":"133","author":[{"given":"Xin","family":"Xiao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daiguo","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiagao","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongchao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,17]]},"reference":[{"key":"2416_CR1","unstructured":"Arpit, D., Jastrzebski, S., Ballas, N., Krueger, D., Bengio, E., Kanwal, M.S., Maharaj, T., Fischer, A., Courville, A., Bengio, Y., et\u00a0al. (2017). A closer look at memorization in deep networks. In: Proceedings of International Conference on Machine Learning, pp. 233\u2013242"},{"key":"2416_CR2","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F. (2022). BEiT: BERT pre-training of image transformers. In: Proceedings of the International Conference on Learning Representations"},{"key":"2416_CR3","doi-asserted-by":"crossref","unstructured":"Borse, S., Wang, Y., Zhang, Y., Porikli, F. (2021). InverseForm: A loss function for structured boundary-aware segmentation. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 5901\u20135911","DOI":"10.1109\/CVPR46437.2021.00584"},{"key":"2416_CR4","doi-asserted-by":"crossref","unstructured":"Cao, Y., Xu, J., Lin, S., Wei, F., Hu, H. (2019). GCNet: Non-local networks meet squeeze-excitation networks and beyond. In: Proceedings of the IEEE International Conference on Computer Vision Workshops","DOI":"10.1109\/ICCVW.2019.00246"},{"key":"2416_CR5","unstructured":"Chatterjee, S., Zielinski, P. (2022). On the generalization mystery in deep learning. arXiv preprint arXiv:2203.10036"},{"key":"2416_CR6","doi-asserted-by":"crossref","unstructured":"Chen, J., Lu, J., Zhu, X., Zhang, L. (2023). Generative semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7111\u20137120","DOI":"10.1109\/CVPR52729.2023.00687"},{"key":"2416_CR7","unstructured":"Chen, L.-C., Papandreou, G., Schroff, F., Adam, H. (2017). Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587"},{"key":"2416_CR8","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H. (2018). Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 801\u2013818","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"2416_CR9","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A.G., Kirillov, A., Girdhar, R. (2022). Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1290\u20131299","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"2416_CR10","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., & Kirillov, A. (2021). Per-pixel classification is not all you need for semantic segmentation. Proceedings of Advances in Neural Information Processing Systems, 34, 17864\u201317875.","journal-title":"Proceedings of Advances in Neural Information Processing Systems"},{"issue":"4","key":"2416_CR11","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"L-C Chen","year":"2017","unstructured":"Chen, L.-C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A. L. (2017). DeepLab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs. IEEE Transactions on Pattern Analysis and Machine Intelligence, 40(4), 834\u2013848.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"1","key":"2416_CR12","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1007\/s10032-019-00334-z","volume":"23","author":"CK Ch\u2019ng","year":"2020","unstructured":"Ch\u2019ng, C. K., Chan, C. S., & Liu, C. (2020). Total-Text: Toward orientation robustness in scene text detection. International Journal on Document Analysis and Recognition, 23(1), 31\u201352.","journal-title":"International Journal on Document Analysis and Recognition"},{"key":"2416_CR13","doi-asserted-by":"crossref","unstructured":"Choi, S., Jung, S., Yun, H., Kim, J.T., Kim, S., Choo, J. (2021). RobustNet: Improving domain generalization in urban-scene segmentation via instance selective whitening. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 11580\u201311590","DOI":"10.1109\/CVPR46437.2021.01141"},{"key":"2416_CR14","doi-asserted-by":"crossref","unstructured":"Chu, J., Chen, Y., Zhou, W., Shi, H., Cao, Y., Tu, D., Jin, R., Xu, Y. (2020). Pay more attention to discontinuity for medical image segmentation. In: Proceedings of International Conference on Medical Image Computing and Computer Assisted Intervention, pp. 166\u2013175","DOI":"10.1007\/978-3-030-59719-1_17"},{"key":"2416_CR15","unstructured":"Contributors, M. (2020). MMSegmentation: OpenMMLab semantic segmentation toolbox and benchmark. https:\/\/github.com\/open-mmlab\/mmsegmentation"},{"key":"2416_CR16","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., Schiele, B. (2016). The Cityscapes dataset for semantic urban scene understanding. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 3213\u20133223","DOI":"10.1109\/CVPR.2016.350"},{"key":"2416_CR17","doi-asserted-by":"crossref","unstructured":"Dai, D., Van\u00a0Gool, L. (2018). Dark model adaptation: Semantic image segmentation from daytime to nighttime. In: 2018 21st International Conference on Intelligent Transportation Systems (ITSC), pp. 3819\u20133824","DOI":"10.1109\/ITSC.2018.8569387"},{"key":"2416_CR18","doi-asserted-by":"crossref","unstructured":"Deng, X., Wang, P., Lian, X., Newsam, S. (2022) NightLab: A dual-level architecture with hardness detection for segmentation at night. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 16938\u201316948","DOI":"10.1109\/CVPR52688.2022.01643"},{"key":"2416_CR19","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In: Proceedings of International Conference on Learning Representations"},{"issue":"7","key":"2416_CR20","first-page":"8577","volume":"45","author":"T Feng","year":"2022","unstructured":"Feng, T., Zhai, Y., Yang, J., Liang, J., Fan, D.-P., Zhang, J., Shao, L., & Tao, D. (2022). Ic9600: A benchmark dataset for automatic image complexity assessment. IEEE Transactions on Pattern Analysis and Machine Intelligent, 45(7), 8577\u20138593.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligent"},{"key":"2416_CR21","doi-asserted-by":"crossref","unstructured":"Fu, J., Liu, J., Tian, H., Li, Y., Bao, Y., Fang, Z., Lu, H. (2019). Dual attention network for scene segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3146\u20133154","DOI":"10.1109\/CVPR.2019.00326"},{"issue":"2","key":"2416_CR22","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TPAMI.2019.2938758","volume":"43","author":"S-H Gao","year":"2019","unstructured":"Gao, S.-H., Cheng, M.-M., Zhao, K., Zhang, X.-Y., Yang, M.-H., & Torr, P. (2019). Res2Net: A new multi-scale backbone architecture. IEEE Transactions on Pattern Analysis and Machine Intelligent, 43(2), 652\u2013662.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligent"},{"key":"2416_CR23","first-page":"1140","volume":"35","author":"M-H Guo","year":"2022","unstructured":"Guo, M.-H., Lu, C.-Z., Hou, Q., Liu, Z., Cheng, M.-M., & Hu, S.-M. (2022). Segnext: Rethinking convolutional attention design for semantic segmentation. Advances in Neural Information Processing Systems, 35, 1140\u20131156.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2416_CR24","doi-asserted-by":"crossref","unstructured":"He, J., Deng, Z., Qiao, Y. (2019). Dynamic multi-scale filters for semantic segmentation. In: Proceedings of IEEE International Conference on Computer Vision, pp. 3562\u20133572","DOI":"10.1109\/ICCV.2019.00366"},{"key":"2416_CR25","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J. (2016). Deep residual learning for image recognition. In: Proceeding of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"2416_CR26","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V., V.\u00a0Le, Q., Adam, H. (2019). Searching for MobileNetV3. In: Proceedings of IEEE International Conference on Computer Vision, pp. 1314\u20131324","DOI":"10.1109\/ICCV.2019.00140"},{"key":"2416_CR27","unstructured":"Howard, A., Zhmoginov, A., Chen, L.-C., Sandler, M., Zhu, M. (2018). Inverted residuals and linear bottlenecks: Mobile networks for classification, detection and segmentation"},{"key":"2416_CR28","unstructured":"Hu, X., Li, F., Samaras, D., Chen, C. (2019). Topology-preserving deep image segmentation. In: Proceedings of the Advances in Neural Information Processing Systems 32"},{"key":"2416_CR29","unstructured":"Hu, X., Wang, Y., Li, F., Samaras, D., Chen, C. (2021). Topology-aware segmentation using discrete morse theory. In: Proceedings of the International Conference on Learning Representations"},{"key":"2416_CR30","doi-asserted-by":"crossref","unstructured":"Huang, J., Guan, D., Xiao, A., Lu, S. (2021). FSDR: Frequency space domain randomization for domain generalization. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 6891\u20136902","DOI":"10.1109\/CVPR46437.2021.00682"},{"key":"2416_CR31","unstructured":"Huang, Z., Wang, X., Wei, Y., Huang, L., Shi, H., Liu, W., Huang, T.S. (2020). CCNet: Criss-cross attention for semantic segmentation. IEEE Transactions on Pattern Analysis and Machanics Intelligent, 1\u20131"},{"key":"2416_CR32","unstructured":"Huang, L., Yuan, Y., Guo, J., Zhang, C., Chen, X., Wang, J. (2019). Interlaced sparse self-attention for semantic segmentation. arXiv preprint arXiv:1907.12273"},{"key":"2416_CR33","doi-asserted-by":"publisher","first-page":"6980","DOI":"10.3390\/s23156980","volume":"23","author":"H Ishikawa","year":"2023","unstructured":"Ishikawa, H., & Yoshimitsu, A. (2023). Boosting semantic segmentation with semantic boundaries. Sensors, 23, 6980. https:\/\/doi.org\/10.3390\/s23156980","journal-title":"Sensors"},{"key":"2416_CR34","doi-asserted-by":"crossref","unstructured":"Jain, J., Singh, A., Orlov, N., Huang, Z., Li, J., Walton, S., Shi, H. (2023). Semask: Semantically masked transformers for semantic segmentation. In: Proceedings of IEEE the Conference on Computer Vision and Pattern Recognition, pp. 752\u2013761","DOI":"10.1109\/ICCVW60793.2023.00083"},{"issue":"6","key":"2416_CR35","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2017). ImageNet classification with deep convolutional neural networks. Communications of the ACM, 60(6), 84\u201390.","journal-title":"Communications of the ACM"},{"key":"2416_CR36","doi-asserted-by":"crossref","unstructured":"Lee, S., Seong, H., Lee, S., Kim, E. (2022). WildNet: Learning domain generalized semantic segmentation from the wild. In: Proceedigs of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9936\u20139946","DOI":"10.1109\/CVPR52688.2022.00970"},{"key":"2416_CR37","doi-asserted-by":"crossref","unstructured":"Li, X., Liu, Z., Luo, P., Change\u00a0Loy, C., Tang, X. (2017). Not all pixels are equal: Difficulty-aware semantic segmentation via deep layer cascade. In: Proceedigs of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3193\u20133202","DOI":"10.1109\/CVPR.2017.684"},{"key":"2416_CR38","doi-asserted-by":"crossref","unstructured":"Li, X., Lv, C., Wang, W., Li, G., Yang, L., Yang, J. (2022). Generalized focal loss: Towards efficient representation learning for dense object detection. IEEE Transactions on Pattern Analysis and Machanics Intelligent, 1\u20131","DOI":"10.1109\/TPAMI.2022.3180392"},{"key":"2416_CR39","unstructured":"Li, Z., Sun, Y., Zhang, L., Tang, J. (2021). CTNet: Context-based tandem network for semantic segmentation. IEEE Transactions on Pattern Analysis and Machanics Intelligent, 1\u20131"},{"key":"2416_CR40","unstructured":"Li, H., Xiong, P., An, J., Wang, L. (2018). Pyramid attention network for semantic segmentation. In: Proceedings of the British Machine Vision Conference"},{"key":"2416_CR41","doi-asserted-by":"crossref","unstructured":"Li, X., Zhong, Z., Wu, J., Yang, Y., Lin, Z., Liu, H. (2019). Expectation-maximization attention networks for semantic segmentation. In: Proceedings of IEEE International Conference on Computer Vision, pp. 9167\u20139176","DOI":"10.1109\/ICCV.2019.00926"},{"key":"2416_CR42","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P. (2017). Focal loss for dense object detection. In: Proceedings of IEEE International Conference on Computer Vision, pp. 2980\u20132988","DOI":"10.1109\/ICCV.2017.324"},{"key":"2416_CR43","unstructured":"Liu, Y., Chen, Y., Lasang, P., Sun, Q. (2022). Covariance attention for semantic segmentation. IEEE Transactions on Pattern Analysis and Machanics Intelligent 44(4), 1805\u20131818"},{"key":"2416_CR44","unstructured":"Liu, J., He, J., Zheng, Y., Yi, S., Wang, X., Li, H. (2021). A holistically-guided decoder for deep representation learning with applications to semantic segmentation and object detection. IEEE Transactions on Pattern Analysis and Machine Intelligent, 1\u20131"},{"key":"2416_CR45","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proc. of IEEE Intl. Conf. on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2416_CR46","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T., Xie, S. (2022). A ConvNet for the 2020s. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 11976\u201311986","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2416_CR47","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T. (2015). Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"7","key":"2416_CR48","first-page":"3523","volume":"44","author":"S Minaee","year":"2022","unstructured":"Minaee, S., Boykov, Y. Y., Porikli, F., Plaza, A. J., Kehtarnavaz, N., & Terzopoulos, D. (2022). Image segmentation using deep learning: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(7), 3523\u20133542.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"10","key":"2416_CR49","doi-asserted-by":"publisher","first-page":"2494","DOI":"10.1007\/s11263-020-01321-2","volume":"128","author":"D Nie","year":"2020","unstructured":"Nie, D., & Shen, D. (2020). Adversarial confidence learning for medical image segmentation and synthesis. International Journal of Computer Vision, 128(10), 2494\u20132513.","journal-title":"International Journal of Computer Vision"},{"key":"2416_CR50","doi-asserted-by":"crossref","unstructured":"Richter, S.R., Vineet, V., Roth, S., Koltun, V. (2016). Playing for data: Ground truth from computer games. In: Proceedings of European Conference on Computer Vision, pp. 102\u2013118","DOI":"10.1007\/978-3-319-46475-6_7"},{"key":"2416_CR51","doi-asserted-by":"crossref","unstructured":"Sakaridis, C., Dai, D., Gool, L.V. (2019). Guided curriculum model adaptation and uncertainty-aware evaluation for semantic nighttime image segmentation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 7374\u20137383","DOI":"10.1109\/ICCV.2019.00747"},{"key":"2416_CR52","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C. (2018). MobileNetV2: Inverted residuals and linear bottlenecks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520","DOI":"10.1109\/CVPR.2018.00474"},{"key":"2416_CR53","doi-asserted-by":"crossref","unstructured":"Shrivastava, A., Gupta, A., Girshick, R. (2016). Training region-based object detectors with online hard example mining. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 761\u2013769","DOI":"10.1109\/CVPR.2016.89"},{"key":"2416_CR54","doi-asserted-by":"crossref","unstructured":"Simonyan, K., Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition. In: Proceedings of the International Conference on Learning Representations","DOI":"10.1109\/ICCV.2015.314"},{"key":"2416_CR55","unstructured":"Tan, M., Le, Q. (2019). EfficientNet: Rethinking model scaling for convolutional neural networks. In: Proceedings of International Conference on Machine Learning, pp. 6105\u20136114"},{"key":"2416_CR56","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2023.106669","volume":"126","author":"H Thisanke","year":"2023","unstructured":"Thisanke, H., Deshan, C., Chamith, K., Seneviratne, S., Vidanaarachchi, R., & Herath, D. (2023). Semantic segmentation using vision transformers: A survey. Engineering Applications of Artificial Intelligence, 126, 106669.","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"2416_CR57","doi-asserted-by":"crossref","unstructured":"Wang, Y., Fei, J., Wang, H., Li, W., Bao, T., Wu, L., Zhao, R., Shen, Y. (2023). Balancing logit variation for long-tailed semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 19561\u201319573","DOI":"10.1109\/CVPR52729.2023.01874"},{"key":"2416_CR58","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K. (2018). Non-local neural networks. In: Proceedings of the European Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803","DOI":"10.1109\/CVPR.2018.00813"},{"key":"2416_CR59","doi-asserted-by":"crossref","unstructured":"Wang, Y., Wang, H., Shen, Y., Fei, J., Li, W., Jin, G., Wu, L., Zhao, R., Le, X. (2022). Semi-supervised semantic segmentation using unreliable pseudo-labels. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 4248\u20134257","DOI":"10.1109\/CVPR52688.2022.00421"},{"key":"2416_CR60","doi-asserted-by":"crossref","unstructured":"Wang, C., Zhang, Y., Cui, M., Liu, J., Ren, P., Yang, Y., Xie, X., Hua, X., Bao, H., Xu, W. (2022). Active boundary loss for semantic segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 2397\u20132405","DOI":"10.1609\/aaai.v36i2.20139"},{"key":"2416_CR61","doi-asserted-by":"crossref","unstructured":"Wang, W., Zhou, T., Yu, F., Dai, J., Konukoglu, E., Van\u00a0Gool, L. (2021). Exploring cross-image pixel contrast for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7303\u20137313","DOI":"10.1109\/ICCV48922.2021.00721"},{"issue":"4","key":"2416_CR62","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A. C., Sheikh, H. R., & Simoncelli, E. P. (2004). Image quality assessment: From error visibility to structural similarity. IEEE Transactions on Image Processing, 13(4), 600\u2013612.","journal-title":"IEEE Transactions on Image Processing"},{"issue":"12","key":"2416_CR63","doi-asserted-by":"publisher","first-page":"3384","DOI":"10.1109\/JBHI.2020.3002985","volume":"24","author":"D Wang","year":"2020","unstructured":"Wang, D., Haytham, A., Pottenburgh, J., Saeedi, O., & Tao, Y. (2020). Hard attention net for automatic retinal vessel segmentation. IEEE Journal of Biomedical and Health Informatics, 24(12), 3384\u20133396.","journal-title":"IEEE Journal of Biomedical and Health Informatics"},{"issue":"10","key":"2416_CR64","doi-asserted-by":"publisher","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","volume":"43","author":"J Wang","year":"2021","unstructured":"Wang, J., Sun, K., Cheng, T., Jiang, B., Deng, C., Zhao, Y., Liu, D., Mu, Y., Tan, M., Wang, X., et al. (2021). Deep high-resolution representation learning for visual recognition. IEEE Transactions on Pattern Analysis and Machine Intelligent, 43(10), 3349\u20133364.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligent"},{"key":"2416_CR65","unstructured":"Waqas\u00a0Zamir, S., Arora, A., Gupta, A., Khan, S., Sun, G., Shahbaz\u00a0Khan, F., Zhu, F., Shao, L., Xia, G.-S., Bai, X. (2019). iSAID: A large-scale dataset for instance segmentation in aerial images. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 28\u201337"},{"key":"2416_CR66","doi-asserted-by":"crossref","unstructured":"Xiao, T., Liu, Y., Zhou, B., Jiang, Y., Sun, J. (2018). Unified perceptual parsing for scene understanding. In: Proceedings of the European Conference on Computer Vision, pp. 418\u2013434","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"2416_CR67","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., He, K. (2017). Aggregated residual transformations for deep neural networks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"key":"2416_CR68","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J. M., & Luo, P. (2021). SegFormer: Simple and efficient design for semantic segmentation with transformers. Proceedings of Advances in Neural Information Processing Systems, 34, 12077\u201312090.","journal-title":"Proceedings of Advances in Neural Information Processing Systems"},{"key":"2416_CR69","doi-asserted-by":"crossref","unstructured":"Xu, J., Xiong, Z., Bhattacharyya, S.P. (2023). PIDNet: A real-time semantic segmentation network inspired by pid controllers. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 19529\u201319539","DOI":"10.1109\/CVPR52729.2023.01871"},{"key":"2416_CR70","doi-asserted-by":"crossref","unstructured":"Yin, M., Yao, Z., Cao, Y., Li, X., Zhang, Z., Lin, S., Hu, H. (2020). Disentangled non-local neural networks. In: Proceedings of the European Conference on Computer Vision, pp. 191\u2013207","DOI":"10.1007\/978-3-030-58555-6_12"},{"issue":"3","key":"2416_CR71","doi-asserted-by":"publisher","first-page":"2665","DOI":"10.1007\/s11063-019-10047-3","volume":"50","author":"J Yin","year":"2019","unstructured":"Yin, J., Xia, P., & He, J. (2019). Online hard region mining for semantic segmentation. Neural Processing Letters, 50(3), 2665\u20132679.","journal-title":"Neural Processing Letters"},{"key":"2416_CR72","doi-asserted-by":"crossref","unstructured":"Yu, F., Chen, H., Wang, X., Xian, W., Chen, Y., Liu, F., Madhavan, V., Darrell, T. (2020). Bdd100k: A diverse driving dataset for heterogeneous multitask learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2636\u20132645","DOI":"10.1109\/CVPR42600.2020.00271"},{"key":"2416_CR73","doi-asserted-by":"crossref","unstructured":"Yu, C., Wang, J., Peng, C., Gao, C., Yu, G., Sang, N. (2018). Learning a discriminative feature network for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1857\u20131866","DOI":"10.1109\/CVPR.2018.00199"},{"key":"2416_CR74","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Chen, X., Wang, J. (2020). Object-contextual representations for semantic segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 173\u2013190","DOI":"10.1007\/978-3-030-58539-6_11"},{"issue":"8","key":"2416_CR75","doi-asserted-by":"publisher","first-page":"2375","DOI":"10.1007\/s11263-021-01465-9","volume":"129","author":"Y Yuan","year":"2021","unstructured":"Yuan, Y., Huang, L., Guo, J., Zhang, C., Chen, X., & Wang, J. (2021). OCNet: Object context for semantic segmentation. International Journal of Computer Vision, 129(8), 2375\u20132398.","journal-title":"International Journal of Computer Vision"},{"issue":"11","key":"2416_CR76","doi-asserted-by":"publisher","first-page":"3051","DOI":"10.1007\/s11263-021-01515-2","volume":"129","author":"C Yu","year":"2021","unstructured":"Yu, C., Gao, C., Wang, J., Yu, G., Shen, C., & Sang, N. (2021). BiSeNet V2: Bilateral network with guided aggregation for real-time semantic segmentation. International Journal of Computer Vision, 129(11), 3051\u20133068.","journal-title":"International Journal of Computer Vision"},{"key":"2416_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z., Sun, Y., He, T., Mueller, J., Manmatha, R., et\u00a0al. (2022). ResNeSt: Split-attention networks. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 2736\u20132746","DOI":"10.1109\/CVPRW56347.2022.00309"},{"key":"2416_CR78","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J. (2017). Pyramid scene parsing network. In: Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, pp. 2881\u20132890","DOI":"10.1109\/CVPR.2017.660"},{"key":"2416_CR79","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Zhong, Z., Zhao, N., Sebe, N., Lee, G.H.: Style-hallucinated dual consistency learning for domain generalized semantic segmentation. In: Proc. of European Conf. on Computer Vision (2022)","DOI":"10.1007\/978-3-031-19815-1_31"},{"key":"2416_CR80","doi-asserted-by":"crossref","unstructured":"Zheng, S., Jayasumana, S., Romera-Paredes, B., Vineet, V., Su, Z., Du, D., Huang, C., Torr, P.H. (2015). Conditional random fields as recurrent neural networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1529\u20131537","DOI":"10.1109\/ICCV.2015.179"},{"key":"2416_CR81","doi-asserted-by":"crossref","unstructured":"Zheng, S., Lu, J., Zhao, H., Zhu, X., Luo, Z., Wang, Y., Fu, Y., Feng, J., Xiang, T., Torr, P.H., et\u00a0al. (2021). Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 6881\u20136890","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"2416_CR82","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Lin, Z.Q., Bidart, R., Hu, X., Daya, I.B., Li, Z., Zheng, W.-S., Li, J., Wong, A. (2020). Squeeze-and-attention networks for semantic segmentation. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 13065\u201313074","DOI":"10.1109\/CVPR42600.2020.01308"},{"issue":"3","key":"2416_CR83","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","volume":"127","author":"B Zhou","year":"2019","unstructured":"Zhou, B., Zhao, H., Puig, X., Xiao, T., Fidler, S., Barriuso, A., & Torralba, A. (2019). Semantic understanding of scenes through the ADE20K dataset. International Journal of Computer Vision, 127(3), 302\u2013321.","journal-title":"International Journal of Computer Vision"},{"key":"2416_CR84","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Xu, M., Bai, S., Huang, T., Bai, X. (2019). Asymmetric non-local neural networks for semantic segmentation. In: Proceedings of IEEE International Conference on Computer Vision, pp. 593\u2013602","DOI":"10.1109\/ICCV.2019.00068"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02416-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02416-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02416-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T05:59:04Z","timestamp":1749275944000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02416-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,17]]},"references-count":84,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["2416"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02416-4","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,17]]},"assertion":[{"value":"26 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}