{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:57:11Z","timestamp":1777654631048,"version":"3.51.4"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730238","type":"print"},{"value":"9783031730245","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73024-5_7","type":"book-chapter","created":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T16:41:03Z","timestamp":1732552863000},"page":"101-117","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["On Learning Discriminative Features from\u00a0Synthesized Data for\u00a0Self-supervised Fine-Grained Visual Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5132-2613","authenticated-orcid":false,"given":"Zihu","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3584-795X","authenticated-orcid":false,"given":"Lingqiao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Scott Ricardo Figueroa","family":"Weston","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Samuel","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3548-4589","authenticated-orcid":false,"given":"Peng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,24]]},"reference":[{"key":"7_CR1","doi-asserted-by":"crossref","unstructured":"Berg, T., Liu, J., Woo\u00a0Lee, S., Alexander, M.L., Jacobs, D.W., Belhumeur, P.N.: Birdsnap: large-scale fine-grained visual categorization of birds. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2011\u20132018 (2014)","DOI":"10.1109\/CVPR.2014.259"},{"key":"7_CR2","unstructured":"Caron, M., Misra, I., Mairal, J., Goyal, P., Bojanowski, P., Joulin, A.: Unsupervised learning of visual features by contrasting cluster assignments. In: Advances in Neural Information Processing Systems, vol. 33, pp. 9912\u20139924 (2020)"},{"key":"7_CR3","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"7_CR4","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"7_CR5","unstructured":"Chen, X., Fan, H., Girshick, R., He, K.: Improved baselines with momentum contrastive learning. arXiv preprint arXiv:2003.04297 (2020)"},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., He, K.: Exploring simple siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15750\u201315758 (2021)","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"7_CR7","unstructured":"Chuang, C.Y., Robinson, J., Lin, Y.C., Torralba, A., Jegelka, S.: Debiased contrastive learning. In: Advances in Neural Information Processing Systems, vol. 33, pp. 8765\u20138775 (2020)"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Codella, N.C., et\u00a0al.: Skin lesion analysis toward melanoma detection: a challenge at the 2017 international symposium on biomedical imaging (ISBI), hosted by the international skin imaging collaboration (ISIC). In: 2018 IEEE 15th International Symposium on Biomedical Imaging (ISBI 2018), pp. 168\u2013172. IEEE (2018)","DOI":"10.1109\/ISBI.2018.8363547"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Cole, E., Yang, X., Wilber, K., Mac\u00a0Aodha, O., Belongie, S.: When does contrastive visual representation learning work? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14755\u201314764 (2022)","DOI":"10.1109\/CVPR52688.2022.01434"},{"key":"7_CR10","unstructured":"Cosentino, R., et al.: Toward a geometrical understanding of self-supervised contrastive learning. arXiv preprint arXiv:2205.06926 (2022)"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: CVPR09 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"7_CR12","unstructured":"Di\u00a0Wu, S.L., et al.: Align yourself: self-supervised pre-training for fine-grained recognition via saliency alignment. arXiv preprint arXiv:2106.15788 (2021)"},{"key":"7_CR13","doi-asserted-by":"crossref","unstructured":"Ericsson, L., Gouk, H., Hospedales, T.M.: How well do self-supervised models transfer? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5414\u20135423 (2021)","DOI":"10.1109\/CVPR46437.2021.00537"},{"key":"7_CR14","doi-asserted-by":"publisher","unstructured":"Gao, Y., et al.: Disco: remedying self-supervised learning on lightweight models with distilled contrastive learning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 237\u2013253. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19809-0_14","DOI":"10.1007\/978-3-031-19809-0_14"},{"key":"7_CR15","unstructured":"Grill, J.B., et al.: Bootstrap your own latent-a new approach to self-supervised learning. In: Advances in Neural Information Processing Systems, vol. 33, pp. 21271\u201321284 (2020)"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9729\u20139738 (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Houben, S., Stallkamp, J., Salmen, J., Schlipsing, M., Igel, C.: Detection of traffic signs in real-world images: the German traffic sign detection benchmark. In: International Joint Conference on Neural Networks, No.\u00a01288 (2013)","DOI":"10.1109\/IJCNN.2013.6706807"},{"key":"7_CR19","doi-asserted-by":"crossref","unstructured":"Hua, T., Wang, W., Xue, Z., Ren, S., Wang, Y., Zhao, H.: On feature decorrelation in self-supervised learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9598\u20139608 (2021)","DOI":"10.1109\/ICCV48922.2021.00946"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Huang, L., You, S., Zheng, M., Wang, F., Qian, C., Yamasaki, T.: Learning where to learn in cross-view self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14451\u201314460 (2022)","DOI":"10.1109\/CVPR52688.2022.01405"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Islam, A., et al.: A broad study on the transferability of visual representations with contrastive learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8845\u20138855 (2021)","DOI":"10.1109\/ICCV48922.2021.00872"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Jang, Y.K., Cho, N.I.: Self-supervised product quantization for deep unsupervised image retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12085\u201312094 (2021)","DOI":"10.1109\/ICCV48922.2021.01187"},{"key":"7_CR23","unstructured":"Jing, L., Vincent, P., LeCun, Y., Tian, Y.: Understanding dimensional collapse in contrastive self-supervised learning. arXiv preprint arXiv:2110.09348 (2021)"},{"key":"7_CR24","doi-asserted-by":"crossref","unstructured":"Kim, S., Bae, S., Yun, S.Y.: Coreset sampling from open-set for fine-grained self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7537\u20137547 (2023)","DOI":"10.1109\/CVPR52729.2023.00728"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J., Fei-Fei, L.: 3D object representations for fine-grained categorization. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp. 554\u2013561 (2013)","DOI":"10.1109\/ICCVW.2013.77"},{"key":"7_CR26","unstructured":"Krizhevsky, A., Hinton, G.: Learning multiple layers of features from tiny images (2009)"},{"key":"7_CR27","unstructured":"Lee, H., Lee, K., Lee, K., Lee, H., Shin, J.: Improving transferability of representations via augmentation-aware self-supervision. In: Advances in Neural Information Processing Systems, vol. 34, pp. 17710\u201317722 (2021)"},{"key":"7_CR28","doi-asserted-by":"publisher","unstructured":"Li, A.C., Efros, A.A., Pathak, D.: Understanding collapse in non-contrastive siamese representation learning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 490\u2013505. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19821-2_28","DOI":"10.1007\/978-3-031-19821-2_28"},{"key":"7_CR29","unstructured":"Maji, S., Rahtu, E., Kannala, J., Blaschko, M., Vedaldi, A.: Fine-grained visual classification of aircraft. arXiv preprint arXiv:1306.5151 (2013)"},{"key":"7_CR30","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"7_CR31","doi-asserted-by":"crossref","unstructured":"Peng, X., Wang, K., Zhu, Z., Wang, M., You, Y.: Crafting better contrastive views for siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16031\u201316040 (2022)","DOI":"10.1109\/CVPR52688.2022.01556"},{"issue":"3","key":"7_CR32","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. Int. J. Comput. Vision 115(3), 211\u2013252 (2015). https:\/\/doi.org\/10.1007\/s11263-015-0816-y","journal-title":"Int. J. Comput. Vision"},{"key":"7_CR33","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-CAM: visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 618\u2013626 (2017)","DOI":"10.1109\/ICCV.2017.74"},{"key":"7_CR34","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Desai, K., Johnson, J., Naik, N.: Casting your model: learning to localize improves self-supervised representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11058\u201311067 (2021)","DOI":"10.1109\/CVPR46437.2021.01091"},{"key":"7_CR35","doi-asserted-by":"crossref","unstructured":"Shu, Y., van\u00a0den Hengel, A., Liu, L.: Learning common rationale to improve self-supervised representation for fine-grained visual recognition problems. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11392\u201311401 (2023)","DOI":"10.1109\/CVPR52729.2023.01096"},{"key":"7_CR36","doi-asserted-by":"publisher","unstructured":"Shu, Y., Yu, B., Xu, H., Liu, L.: Improving fine-grained visual recognition in low data regimes via self-boosting attention mechanism. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 449\u2013465. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19806-9_26","DOI":"10.1007\/978-3-031-19806-9_26"},{"issue":"2","key":"7_CR37","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1145\/2812802","volume":"59","author":"B Thomee","year":"2016","unstructured":"Thomee, B., et al.: YFCC100M: the new data in multimedia research. Commun. ACM 59(2), 64\u201373 (2016)","journal-title":"Commun. ACM"},{"key":"7_CR38","unstructured":"Wah, C., Branson, S., Welinder, P., Perona, P., Belongie, S.: The caltech-UCSD birds-200-2011 dataset (2011)"},{"key":"7_CR39","unstructured":"Wang, T., Isola, P.: Understanding contrastive representation learning through alignment and uniformity on the hypersphere. In: International Conference on Machine Learning, pp. 9929\u20139939. PMLR (2020)"},{"key":"7_CR40","unstructured":"Wang, Z., Wang, Y., Hu, H., Li, P.: Contrastive learning with consistent representations. arXiv preprint arXiv:2302.01541 (2023)"},{"key":"7_CR41","unstructured":"Xiao, T., Wang, X., Efros, A.A., Darrell, T.: What should not be contrastive in contrastive learning. arXiv preprint arXiv:2008.05659 (2020)"},{"key":"7_CR42","doi-asserted-by":"crossref","unstructured":"Yao, Y., Ye, C., He, J., Elsayed, G.F.: Teacher-generated spatial-attention labels boost robustness and accuracy of contrastive models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23282\u201323291 (2023)","DOI":"10.1109\/CVPR52729.2023.02230"},{"key":"7_CR43","doi-asserted-by":"crossref","unstructured":"Yeh, C.H., Hong, C.Y., Hsu, Y.C., Liu, T.L.: Saga: self-augmentation with guided attention for representation learning. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 3463\u20133467. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747302"},{"key":"7_CR44","doi-asserted-by":"crossref","unstructured":"Zhao, N., Wu, Z., Lau, R.W., Lin, S.: Distilling localization for self-supervised representation learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 10990\u201310998 (2021)","DOI":"10.1609\/aaai.v35i12.17312"},{"key":"7_CR45","unstructured":"Ziyin, L., Lubana, E.S., Ueda, M., Tanaka, H.: What shapes the loss landscape of self-supervised learning? arXiv preprint arXiv:2210.00638 (2022)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73024-5_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,25]],"date-time":"2024-11-25T17:06:05Z","timestamp":1732554365000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73024-5_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,24]]},"ISBN":["9783031730238","9783031730245"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73024-5_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,24]]},"assertion":[{"value":"24 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}