{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:01:45Z","timestamp":1777654905402,"version":"3.51.4"},"publisher-location":"Cham","reference-count":53,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726422","type":"print"},{"value":"9783031726439","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72643-9_21","type":"book-chapter","created":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T20:48:33Z","timestamp":1732222113000},"page":"352-369","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Learning from\u00a0the\u00a0Web: Language Drives Weakly-Supervised Incremental Learning for\u00a0Semantic Segmentation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5321-2264","authenticated-orcid":false,"given":"Chang","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1390-8419","authenticated-orcid":false,"given":"Giulia","family":"Rizzoli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9502-2389","authenticated-orcid":false,"given":"Pietro","family":"Zanuttigh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0319-0308","authenticated-orcid":false,"given":"Fu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7359-276X","authenticated-orcid":false,"given":"Yi","family":"Niu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,22]]},"reference":[{"key":"21_CR1","doi-asserted-by":"crossref","unstructured":"Ahn, J., Kwak, S.: Learning pixel-level semantic affinity with image-level supervision for weakly supervised semantic segmentation. In: Proceedings of the IEEE Conference On Computer Vision And Pattern Recognition, pp. 4981\u20134990 (2018)","DOI":"10.1109\/CVPR.2018.00523"},{"key":"21_CR2","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. ArXiv:abs\/2204.14198 (2022)"},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Araslanov, N., Roth, S.: Single-stage semantic segmentation from image labels. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4253\u20134262 (2020)","DOI":"10.1109\/CVPR42600.2020.00431"},{"key":"21_CR4","unstructured":"Awadalla, A., et al.: Openflamingo: An open-source framework for training large autoregressive vision-language models. arXiv preprint arXiv:2308.01390 (2023)"},{"key":"21_CR5","unstructured":"Bird, S., Klein, E., Loper, E.: Natural language processing with Python: analyzing text with the natural language toolkit. \" O\u2019Reilly Media, Inc.\" (2009)"},{"key":"21_CR6","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Cermelli, F., Fontanel, D., Tavera, A., Ciccone, M., Caputo, B.: Incremental learning in semantic segmentation from image labels. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4371\u20134381 (2022)","DOI":"10.1109\/CVPR52688.2022.00433"},{"key":"21_CR8","doi-asserted-by":"crossref","unstructured":"Cermelli, F., Mancini, M., Bul\u00f2, S.R., Ricci, E., Caputo, B.: Modeling the background for incremental learning in semantic segmentation. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00925"},{"issue":"4","key":"21_CR9","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2017","unstructured":"Chen, L.C., Papandreou, G., Kokkinos, I., Murphy, K., Yuille, A.L.: sDeeplab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans. Pattern Anal. Mach. Intell. 40(4), 834\u2013848 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"21_CR10","doi-asserted-by":"crossref","unstructured":"Chen, X., Gupta, A.: Webly supervised learning of convolutional networks. In: ICCV, pp. 1431\u20131439 (2015)","DOI":"10.1109\/ICCV.2015.168"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Dai, J., He, K., Sun, J.: Boxsup: Exploiting bounding boxes to supervise convolutional networks for semantic segmentation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1635\u20131643 (2015)","DOI":"10.1109\/ICCV.2015.191"},{"key":"21_CR12","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"21_CR13","doi-asserted-by":"crossref","unstructured":"Douillard, A., Chen, Y., Dapogny, A., Cord, M.: Plop: Learning without forgetting for continual semantic segmentation. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00403"},{"key":"21_CR14","doi-asserted-by":"publisher","first-page":"670","DOI":"10.1007\/978-3-030-58555-6_40","volume-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV","author":"H Duan","year":"2020","unstructured":"Duan, H., Zhao, Y., Xiong, Y., Liu, W., Lin, D.: Omni-sourced webly-supervised learning for video recognition. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV, pp. 670\u2013688. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58555-6_40"},{"key":"21_CR15","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C.K., Winn, J., Zisserman, A.: The pascal visual object classes (voc) challenge. Int. J. Comput. Vision 88, 303\u2013338 (2010)","journal-title":"Int. J. Comput. Vision"},{"key":"21_CR16","doi-asserted-by":"crossref","unstructured":"Fellbaum, C.: WordNet: An electronic lexical database. MIT press (1998)","DOI":"10.7551\/mitpress\/7287.001.0001"},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"21_CR18","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Klingner, M., B\u00e4r, A., Donn, P., Fingscheidt, T.: Class-incremental learning for semantic segmentation re-using neither old data nor old labels. International Conference on Intelligent Transportation Systems (2020)","DOI":"10.1109\/ITSC45102.2020.9294483"},{"key":"21_CR20","doi-asserted-by":"crossref","unstructured":"Lee, S., Lee, M., Lee, J., Shim, H.: Railroad is not a train: saliency as pseudo-pixel supervision for weakly supervised semantic segmentation. In: Proceedings of the IEEE\/Cvf Conference on Computer Vision and Pattern Recognition, pp. 5495\u20135505 (2021)","DOI":"10.1109\/CVPR46437.2021.00545"},{"key":"21_CR21","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Gligen: Open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22511\u201322521 (2023)","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"21_CR22","doi-asserted-by":"crossref","unstructured":"Lin, D., Dai, J., Jia, J., He, K., Sun, J.: Scribblesup: scribble-supervised convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3159\u20133167 (2016)","DOI":"10.1109\/CVPR.2016.344"},{"key":"21_CR23","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision \u2013 ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V, pp. 740\u2013755. Springer International Publishing, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"21_CR24","unstructured":"Liu, C., Rizzoli, G., Barbato, F., Michieli, U., Niu, Y., Zanuttigh, P.: Recall+: Adversarial web-based replay for continual learning in semantic segmentation. arXiv preprint arXiv:2309.10479 (2023)"},{"key":"21_CR25","unstructured":"Lukasik, M., Bhojanapalli, S., Menon, A., Kumar, S.: Does label smoothing mitigate label noise? In: International Conference on Machine Learning, pp. 6448\u20136458. PMLR (2020)"},{"key":"21_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107308","volume":"103","author":"A Luo","year":"2020","unstructured":"Luo, A., Li, X., Yang, F., Jiao, Z., Cheng, H.: Webly-supervised learning for salient object detection. Pattern Recogn. 103, 107308 (2020)","journal-title":"Pattern Recogn."},{"key":"21_CR27","doi-asserted-by":"crossref","unstructured":"Maracani, A., Michieli, U., Toldo, M., Zanuttigh, P.: Recall: Replay-based continual learning in semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7026\u20137035 (2021)","DOI":"10.1109\/ICCV48922.2021.00694"},{"key":"21_CR28","unstructured":"McEver, R.A., Manjunath, B.: Pcams: Weakly supervised semantic segmentation using point supervision. arXiv preprint arXiv:2007.05615 (2020)"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Michieli, U., Zanuttigh, P.: Incremental Learning Techniques for Semantic Segmentation (2019)","DOI":"10.1109\/ICCVW.2019.00400"},{"key":"21_CR30","doi-asserted-by":"crossref","unstructured":"Michieli, U., Zanuttigh, P.: Continual semantic segmentation via repulsion-attraction of sparse and disentangled latent representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1114\u20131124 (2021)","DOI":"10.1109\/CVPR46437.2021.00117"},{"key":"21_CR31","doi-asserted-by":"crossref","unstructured":"Niu, L., Veeraraghavan, A., Sabharwal, A.: Webly supervised learning meets zero-shot learning: A hybrid approach for fine-grained classification. In: CVPR, pp. 7171\u20137180 (2018)","DOI":"10.1109\/CVPR.2018.00749"},{"key":"21_CR32","doi-asserted-by":"crossref","unstructured":"Phan, M.H., Phung, S.L., Tran-Thanh, L., Bouzerdoum, A., et\u00a0al.: Class similarity weighted knowledge distillation for continual semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16866\u201316875 (2022)","DOI":"10.1109\/CVPR52688.2022.01636"},{"key":"21_CR33","unstructured":"Qin, Z., et al.: Dataset growth (2024). https:\/\/arxiv.org\/abs\/2405.18347"},{"key":"21_CR34","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR (18\u201324 Jul 2021), https:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"21_CR35","doi-asserted-by":"crossref","unstructured":"Ravi, S., Chinchure, A., Sigal, L., Liao, R., Shwartz, V.: Vlc-bert: visual question answering with contextualized commonsense knowledge. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1155\u20131165 (2023)","DOI":"10.1109\/WACV56688.2023.00121"},{"key":"21_CR36","doi-asserted-by":"crossref","unstructured":"Rizzoli, G., Shenaj, D., Zanuttigh, P.: Source-free domain adaptation for rgb-d semantic segmentation with vision transformers. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 615\u2013624 (2024)","DOI":"10.1109\/WACVW60836.2024.00070"},{"key":"21_CR37","unstructured":"Roy, S., Volpi, R., Csurka, G., Larlus, D.: Rasp: Relation-aware semantic prior for weakly supervised incremental segmentation. arXiv preprint arXiv:2305.19879 (2023)"},{"key":"21_CR38","doi-asserted-by":"crossref","unstructured":"Shenaj, D., Rizzoli, G., Zanuttigh, P.: Federated learning in computer vision. IEEE Access (2023)","DOI":"10.1109\/ACCESS.2023.3310400"},{"key":"21_CR39","doi-asserted-by":"crossref","unstructured":"Song, C., Huang, Y., Ouyang, W., Wang, L.: Box-driven class-wise region masking and filling rate guided loss for weakly supervised semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3136\u20133145 (2019)","DOI":"10.1109\/CVPR.2019.00325"},{"key":"21_CR40","doi-asserted-by":"crossref","unstructured":"Sun, H., He, X., Peng, Y.: Hcl: Hierarchical consistency learning for webly supervised fine-grained recognition. IEEE Transactions on Multimedia (2023)","DOI":"10.1109\/TMM.2023.3330076"},{"key":"21_CR41","unstructured":"Tan, M., Le, Q.: Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114. PMLR (2019)"},{"key":"21_CR42","doi-asserted-by":"crossref","unstructured":"Taylor, A., Marcus, M., Santorini, B.: The penn treebank: an overview. Treebanks: Building and using parsed corpora, pp. 5\u201322 (2003)","DOI":"10.1007\/978-94-010-0201-1_1"},{"key":"21_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Y., Zhang, J., Kan, M., Shan, S., Chen, X.: Self-supervised equivariant attention mechanism for weakly supervised semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12275\u201312284 (2020)","DOI":"10.1109\/CVPR42600.2020.01229"},{"key":"21_CR44","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1016\/j.patcog.2019.01.006","volume":"90","author":"Z Wu","year":"2019","unstructured":"Wu, Z., Shen, C., Van Den Hengel, A.: Wider or deeper: revisiting the resnet model for visual recognition. Pattern Recogn. 90, 119\u2013133 (2019)","journal-title":"Pattern Recogn."},{"key":"21_CR45","doi-asserted-by":"crossref","unstructured":"Xiao, J.W., Zhang, C.B., Feng, J., Liu, X., van\u00a0de Weijer, J., Cheng, M.M.: Endpoints weight fusion for class incremental semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7204\u20137213 (2023)","DOI":"10.1109\/CVPR52729.2023.00696"},{"key":"21_CR46","unstructured":"Yang, G., et al.: Uncertainty-aware contrastive distillation for incremental semantic segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)"},{"key":"21_CR47","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"779","DOI":"10.1007\/978-3-030-58598-3_46","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Yang","year":"2020","unstructured":"Yang, J., et al.: Webly supervised image classification with self-contained confidence. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12353, pp. 779\u2013795. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58598-3_46"},{"key":"21_CR48","doi-asserted-by":"crossref","unstructured":"Yang, Y., Soatto, S.: Fda: Fourier domain adaptation for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4085\u20134095 (2020)","DOI":"10.1109\/CVPR42600.2020.00414"},{"key":"21_CR49","doi-asserted-by":"crossref","unstructured":"Yu, C., Zhou, Q., Li, J., Yuan, J., Wang, Z., Wang, F.: Foundation model drives weakly incremental learning for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23685\u201323694 (2023)","DOI":"10.1109\/CVPR52729.2023.02268"},{"key":"21_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, C.B., Xiao, J.W., Liu, X., Chen, Y.C., Cheng, M.M.: Representation compensation networks for continual semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7053\u20137064 (2022)","DOI":"10.1109\/CVPR52688.2022.00692"},{"key":"21_CR51","doi-asserted-by":"publisher","unstructured":"Zhao, H., Yang, F., Fu, X., Li, X.: Rbc: Rectifying the biased context in continual semantic segmentation. In: European Conference on Computer Vision, pp. 55\u201372. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19830-4_4","DOI":"10.1007\/978-3-031-19830-4_4"},{"key":"21_CR52","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., Torralba, A.: Learning deep features for discriminative localization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2921\u20132929 (2016)","DOI":"10.1109\/CVPR.2016.319"},{"key":"21_CR53","doi-asserted-by":"publisher","unstructured":"Zhou, C., Loy, C.C., Dai, B.: Extract free dense labels from CLIP. In: European Conference on Computer Vision, pp. 696\u2013712. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_40","DOI":"10.1007\/978-3-031-19815-1_40"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72643-9_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T21:27:54Z","timestamp":1732224474000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72643-9_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,22]]},"ISBN":["9783031726422","9783031726439"],"references-count":53,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72643-9_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,22]]},"assertion":[{"value":"22 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}