{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T23:09:33Z","timestamp":1778800173257,"version":"3.51.4"},"reference-count":120,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T00:00:00Z","timestamp":1776902400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100004794","name":"Centre National de la Recherche Scientifique","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004794","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003990","name":"Conseil R\u00e9gional, \u00cele-de-France","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003990","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.cviu.2026.104783","type":"journal-article","created":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T15:43:00Z","timestamp":1777131780000},"page":"104783","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Self-supervised learning for object detection in challenging settings: A survey"],"prefix":"10.1016","volume":"268","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5145-6938","authenticated-orcid":false,"given":"Alina","family":"Ciocarlan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8332-972X","authenticated-orcid":false,"given":"Sidonie","family":"Lefebvre","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8494-2289","authenticated-orcid":false,"given":"Sylvie","family":"Le H\u00e9garat-Mascle","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1987-1913","authenticated-orcid":false,"given":"Arnaud","family":"Woiselle","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104783_b1","unstructured":"Asano,\u00a0Y., Rupprecht,\u00a0C., Vedaldi,\u00a0A., 2019. Self-labelling via simultaneous clustering and representation learning. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b2","series-title":"European Conference on Computer Vision","first-page":"456","article-title":"Masked siamese networks for label-efficient learning","author":"Assran","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b3","doi-asserted-by":"crossref","unstructured":"Assran,\u00a0M., Duval,\u00a0Q., Misra,\u00a0I., Bojanowski,\u00a0P., Vincent,\u00a0P., Rabbat,\u00a0M., LeCun,\u00a0Y., Ballas,\u00a0N., 2023. Self-supervised learning from images with a joint-embedding predictive architecture. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 15619\u201315629.","DOI":"10.1109\/CVPR52729.2023.01499"},{"key":"10.1016\/j.cviu.2026.104783_b4","unstructured":"Bao,\u00a0H., Dong,\u00a0L., Piao,\u00a0S., Wei,\u00a0F., 2021. BEiT: BERT Pre-Training of Image Transformers. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b5","unstructured":"Bardes,\u00a0A., Ponce,\u00a0J., Lecun,\u00a0Y., 2022a. VICReg: Variance-Invariance-Covariance Regularization For Self-Supervised Learning. In: ICLR 2022-International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b6","doi-asserted-by":"crossref","first-page":"8799","DOI":"10.52202\/068431-0640","article-title":"Vicregl: Self-supervised learning of local visual features","volume":"35","author":"Bardes","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b7","doi-asserted-by":"crossref","unstructured":"Caron,\u00a0M., Bojanowski,\u00a0P., Joulin,\u00a0A., Douze,\u00a0M., 2018. Deep clustering for unsupervised learning of visual features. In: Proceedings of the European Conference on Computer Vision. ECCV, pp. 132\u2013149.","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"10.1016\/j.cviu.2026.104783_b8","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume":"33","author":"Caron","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b9","doi-asserted-by":"crossref","unstructured":"Caron,\u00a0M., Touvron,\u00a0H., Misra,\u00a0I., J\u00e9gou,\u00a0H., Mairal,\u00a0J., Bojanowski,\u00a0P., Joulin,\u00a0A., 2021. Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9650\u20139660.","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"10.1016\/j.cviu.2026.104783_b10","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0X., He,\u00a0K., 2021. Exploring simple siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 15750\u201315758.","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"10.1016\/j.cviu.2026.104783_b11","series-title":"International Conference on Machine Learning","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","author":"Chen","year":"2020"},{"key":"10.1016\/j.cviu.2026.104783_b12","doi-asserted-by":"crossref","unstructured":"Chen,\u00a0X., Xie,\u00a0S., He,\u00a0K., 2021. An empirical study of training self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9640\u20139649.","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"10.1016\/j.cviu.2026.104783_b13","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2023.3290594","article-title":"Towards large-scale small object detection: Survey and benchmarks","author":"Cheng","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104783_b14","doi-asserted-by":"crossref","unstructured":"Chin,\u00a0Z.-Y., Jiang,\u00a0C.-M., Huang,\u00a0C.-C., Chen,\u00a0P.-Y., Chiu,\u00a0W.-C., 2024. Masking Improves Contrastive Self-Supervised Learning for ConvNets, and Saliency Tells You Where. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2761\u20132770.","DOI":"10.1109\/WACV57701.2024.00274"},{"key":"10.1016\/j.cviu.2026.104783_b15","first-page":"8765","article-title":"Debiased contrastive learning","volume":"33","author":"Chuang","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b16","doi-asserted-by":"crossref","unstructured":"Cubuk,\u00a0E.D., Zoph,\u00a0B., Mane,\u00a0D., Vasudevan,\u00a0V., Le,\u00a0Q.V., 2019. Autoaugment: Learning augmentation strategies from data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 113\u2013123.","DOI":"10.1109\/CVPR.2019.00020"},{"key":"10.1016\/j.cviu.2026.104783_b17","series-title":"2009 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"248","article-title":"Imagenet: A large-scale hierarchical image database","author":"Deng","year":"2009"},{"key":"10.1016\/j.cviu.2026.104783_b18","series-title":"Improved regularization of convolutional neural networks with cutout","author":"DeVries","year":"2017"},{"key":"10.1016\/j.cviu.2026.104783_b19","article-title":"Deeply unsupervised patch re-identification for pre-training object detectors","author":"Ding","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104783_b20","series-title":"Are large-scale datasets necessary for self-supervised pre-training?","author":"El-Nouby","year":"2021"},{"key":"10.1016\/j.cviu.2026.104783_b21","series-title":"International Conference on Machine Learning","first-page":"3015","article-title":"Whitening for self-supervised representation learning","author":"Ermolov","year":"2021"},{"key":"10.1016\/j.cviu.2026.104783_b22","unstructured":"Fang,\u00a0Y., Dong,\u00a0L., Bao,\u00a0H., Wang,\u00a0X., Wei,\u00a0F., 2022. Corrupted Image Modeling for Self-Supervised Visual Pre-Training. In: The Eleventh International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b23","doi-asserted-by":"crossref","unstructured":"Fang,\u00a0Y., Yang,\u00a0S., Wang,\u00a0S., Ge,\u00a0Y., Shan,\u00a0Y., Wang,\u00a0X., 2023. Unleashing vanilla vision transformer with masked image modeling for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6244\u20136253.","DOI":"10.1109\/ICCV51070.2023.00574"},{"key":"10.1016\/j.cviu.2026.104783_b24","doi-asserted-by":"crossref","first-page":"167","DOI":"10.1023\/B:VISI.0000022288.19776.77","article-title":"Efficient graph-based image segmentation","volume":"59","author":"Felzenszwalb","year":"2004","journal-title":"Int. J. Comput. Vis."},{"issue":"7","key":"10.1016\/j.cviu.2026.104783_b25","doi-asserted-by":"crossref","first-page":"3918","DOI":"10.1007\/s11263-024-02327-w","article-title":"An experimental study on exploring strong lightweight vision transformers via masked image modeling pre-training","volume":"133","author":"Gao","year":"2025","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.cviu.2026.104783_b26","series-title":"Convmae: Masked convolution meets masked autoencoders","author":"Gao","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b27","first-page":"1","article-title":"BS 3 LNet: A new blind-spot self-supervised learning network for hyperspectral anomaly detection","volume":"61","author":"Gao","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.cviu.2026.104783_b28","first-page":"23885","article-title":"Partial success in closing the gap between human and machine vision","volume":"34","author":"Geirhos","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b29","unstructured":"Geirhos,\u00a0R., Rubisch,\u00a0P., Michaelis,\u00a0C., Bethge,\u00a0M., Wichmann,\u00a0F.A., Brendel,\u00a0W., 2018. ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness. In: International conference on learning representations."},{"key":"10.1016\/j.cviu.2026.104783_b30","series-title":"Accurate, large minibatch sgd: Training imagenet in 1 hour","author":"Goyal","year":"2017"},{"key":"10.1016\/j.cviu.2026.104783_b31","first-page":"21271","article-title":"Bootstrap your own latent-a new approach to self-supervised learning","volume":"33","author":"Grill","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b32","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2024.3415112","article-title":"A survey on self-supervised learning: Algorithms, applications, and future trends","author":"Gui","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104783_b33","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Chen,\u00a0X., Xie,\u00a0S., Li,\u00a0Y., Doll\u00e1r,\u00a0P., Girshick,\u00a0R., 2022. Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16000\u201316009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"10.1016\/j.cviu.2026.104783_b34","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Fan,\u00a0H., Wu,\u00a0Y., Xie,\u00a0S., Girshick,\u00a0R., 2020. Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 9729\u20139738.","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"10.1016\/j.cviu.2026.104783_b35","doi-asserted-by":"crossref","unstructured":"He,\u00a0K., Gkioxari,\u00a0G., Doll\u00e1r,\u00a0P., Girshick,\u00a0R., 2017. Mask r-cnn. In: Proceedings of the IEEE international conference on computer vision. pp. 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"10.1016\/j.cviu.2026.104783_b36","doi-asserted-by":"crossref","unstructured":"H\u00e9naff,\u00a0O.J., Koppula,\u00a0S., Alayrac,\u00a0J.-B., Van\u00a0den Oord,\u00a0A., Vinyals,\u00a0O., Carreira,\u00a0J., 2021. Efficient visual pretraining with contrastive detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10086\u201310096.","DOI":"10.1109\/ICCV48922.2021.00993"},{"key":"10.1016\/j.cviu.2026.104783_b37","series-title":"European Conference on Computer Vision","first-page":"123","article-title":"Object discovery and representation networks","author":"H\u00e9naff","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b38","series-title":"European Conference on Computer Vision","first-page":"444","article-title":"Vic-mae: Self-supervised representation learning from images and video with contrastive masked autoencoders","author":"Hernandez","year":"2024"},{"key":"10.1016\/j.cviu.2026.104783_b39","series-title":"Milan: Masked image pretraining on language assisted representation","author":"Hou","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b40","article-title":"Contrastive masked autoencoders are stronger vision learners","author":"Huang","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"10.1016\/j.cviu.2026.104783_b41","first-page":"4071","article-title":"A survey of self-supervised and few-shot object detection","volume":"45","author":"Huang","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104783_b42","article-title":"Self-supervised masking for unsupervised anomaly detection and localization","author":"Huang","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.cviu.2026.104783_b43","series-title":"FLIR data set dataset","author":"Imaging","year":"2024"},{"key":"10.1016\/j.cviu.2026.104783_b44","doi-asserted-by":"crossref","unstructured":"Islam,\u00a0A., Lundell,\u00a0B., Sawhney,\u00a0H., Sinha,\u00a0S.N., Morales,\u00a0P., Radke,\u00a0R.J., 2023. Self-supervised Learning with Local Contrastive Loss for Detection and Semantic Segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 5624\u20135633.","DOI":"10.1109\/WACV56688.2023.00558"},{"issue":"1","key":"10.1016\/j.cviu.2026.104783_b45","doi-asserted-by":"crossref","first-page":"2","DOI":"10.3390\/technologies9010002","article-title":"A survey on contrastive self-supervised learning","volume":"9","author":"Jaiswal","year":"2020","journal-title":"Technologies"},{"key":"10.1016\/j.cviu.2026.104783_b46","series-title":"European Conference on Computer Vision","first-page":"300","article-title":"What to hide from your students: Attention-guided masked image modeling","author":"Kakogeorgiou","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b47","series-title":"2024 International Conference on Machine Learning and Applications (ICMLA)","first-page":"1111","article-title":"Cnn-jepa: self-supervised pretraining convolutional neural networks using joint embedding predictive architecture","author":"Kalapos","year":"2024"},{"key":"10.1016\/j.cviu.2026.104783_b48","series-title":"A survey of the self supervised learning mechanisms for vision transformers","author":"Khan","year":"2024"},{"key":"10.1016\/j.cviu.2026.104783_b49","first-page":"13165","article-title":"Mst: Masked self-supervised transformer for visual representation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b50","first-page":"1","article-title":"Global and local contrastive self-supervised learning for semantic segmentation of HR remote sensing images","volume":"60","author":"Li","year":"2022","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.cviu.2026.104783_b51","first-page":"1","article-title":"You only train once: Learning a general anomaly enhancement network with random masks for hyperspectral anomaly detection","volume":"61","author":"Li","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.cviu.2026.104783_b52","series-title":"International Conference on Machine Learning","first-page":"20149","article-title":"Architecture-agnostic masked image modeling from ViT back to CNN","author":"Li","year":"2023"},{"key":"10.1016\/j.cviu.2026.104783_b53","series-title":"Benchmarking detection transfer learning with vision transformers","author":"Li","year":"2021"},{"key":"10.1016\/j.cviu.2026.104783_b54","doi-asserted-by":"crossref","first-page":"14290","DOI":"10.52202\/068431-1039","article-title":"Semmae: Semantic-guided masking for learning masked autoencoders","volume":"35","author":"Li","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b55","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.cviu.2026.104783_b56","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Z., Gui,\u00a0J., Luo,\u00a0H., 2023. Good helper is around you: Attention-driven masked image modeling. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 37, pp. 1799\u20131807.","DOI":"10.1609\/aaai.v37i2.25269"},{"key":"10.1016\/j.cviu.2026.104783_b57","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0J., Huang,\u00a0X., Zheng,\u00a0J., Liu,\u00a0Y., Li,\u00a0H., 2023. MixMAE: Mixed and masked autoencoder for efficient pretraining of hierarchical vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6252\u20136261.","DOI":"10.1109\/CVPR52729.2023.00605"},{"key":"10.1016\/j.cviu.2026.104783_b58","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0H., Jiang,\u00a0X., Li,\u00a0X., Guo,\u00a0A., Hu,\u00a0Y., Jiang,\u00a0D., Ren,\u00a0B., 2023. The devil is in the frequency: Geminated gestalt autoencoder for self-supervised visual pre-training. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 37, pp. 1649\u20131656.","DOI":"10.1609\/aaai.v37i2.25252"},{"key":"10.1016\/j.cviu.2026.104783_b59","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Q., Li,\u00a0X., He,\u00a0Z., Li,\u00a0C., Li,\u00a0J., Zhou,\u00a0Z., Yuan,\u00a0D., Li,\u00a0J., Yang,\u00a0K., Fan,\u00a0N., et al., 2020. LSOTB-TIR: A large-scale high-diversity thermal infrared object tracking benchmark. In: Proceedings of the 28th ACM International Conference on Multimedia. pp. 3847\u20133856.","DOI":"10.1145\/3394171.3413922"},{"key":"10.1016\/j.cviu.2026.104783_b60","series-title":"Self-emd: Self-supervised object detection without imagenet","author":"Liu","year":"2020"},{"key":"10.1016\/j.cviu.2026.104783_b61","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Z., Lin,\u00a0Y., Cao,\u00a0Y., Hu,\u00a0H., Wei,\u00a0Y., Zhang,\u00a0Z., Lin,\u00a0S., Guo,\u00a0B., 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.cviu.2026.104783_b62","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0Z., Mao,\u00a0H., Wu,\u00a0C.-Y., Feichtenhofer,\u00a0C., Darrell,\u00a0T., Xie,\u00a0S., 2022. A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11976\u201311986.","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"10.1016\/j.cviu.2026.104783_b63","article-title":"PixMIM: Rethinking pixel reconstruction in masked image modeling","author":"Liu","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.cviu.2026.104783_b64","series-title":"Exploring target representations for masked autoencoders","author":"Liu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b65","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.cviu.2026.104783_b66","doi-asserted-by":"crossref","unstructured":"Misra,\u00a0I., Maaten,\u00a0L.v.d., 2020. Self-supervised learning of pretext-invariant representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6707\u20136717.","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"10.1016\/j.cviu.2026.104783_b67","unstructured":"Mitrovic,\u00a0J., McWilliams,\u00a0B., Walker,\u00a0J.C., Buesing,\u00a0L.H., Blundell,\u00a0C., 2020. Representation Learning via Invariant Causal Mechanisms. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b68","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TGRS.2023.3268232","article-title":"Cmid: A unified self-supervised learning framework for remote sensing image understanding","volume":"61","author":"Muhtar","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.cviu.2026.104783_b69","first-page":"4489","article-title":"Unsupervised learning of dense visual representations","volume":"33","author":"O\u00a0Pinheiro","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b70","series-title":"Dinov2: Learning robust visual features without supervision","author":"Oquab","year":"2023"},{"key":"10.1016\/j.cviu.2026.104783_b71","series-title":"Know your self-supervised learning: A survey on image-based generative and discriminative training","author":"Ozbulak","year":"2023"},{"key":"10.1016\/j.cviu.2026.104783_b72","unstructured":"Park,\u00a0N., Kim,\u00a0W., Heo,\u00a0B., Kim,\u00a0T., Yun,\u00a0S., 2022. What Do Self-Supervised Vision Transformers Learn?. In: The Eleventh International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b73","doi-asserted-by":"crossref","unstructured":"Pathak,\u00a0D., Krahenbuhl,\u00a0P., Donahue,\u00a0J., Darrell,\u00a0T., Efros,\u00a0A.A., 2016. Context encoders: Feature learning by inpainting. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 2536\u20132544.","DOI":"10.1109\/CVPR.2016.278"},{"key":"10.1016\/j.cviu.2026.104783_b74","article-title":"A unified view of masked image modeling","author":"Peng","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.cviu.2026.104783_b75","doi-asserted-by":"crossref","unstructured":"Peng,\u00a0X., Wang,\u00a0K., Zhu,\u00a0Z., Wang,\u00a0M., You,\u00a0Y., 2022. Crafting better contrastive views for siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16031\u201316040.","DOI":"10.1109\/CVPR52688.2022.01556"},{"key":"10.1016\/j.cviu.2026.104783_b76","series-title":"2014 IEEE International Conference on Robotics and Automation","first-page":"1794","article-title":"People detection and tracking from aerial thermal views","author":"Portmann","year":"2014"},{"key":"10.1016\/j.cviu.2026.104783_b77","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cviu.2026.104783_b78","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1016\/j.jvcir.2015.11.002","article-title":"Vehicle detection in aerial imagery: A small target detection benchmark","volume":"34","author":"Razakarivony","year":"2016","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.cviu.2026.104783_b79","doi-asserted-by":"crossref","unstructured":"Reed,\u00a0C.J., Metzger,\u00a0S., Srinivas,\u00a0A., Darrell,\u00a0T., Keutzer,\u00a0K., 2021. Selfaugment: Automatic augmentation policies for self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2674\u20132683.","DOI":"10.1109\/CVPR46437.2021.00270"},{"key":"10.1016\/j.cviu.2026.104783_b80","doi-asserted-by":"crossref","unstructured":"Roh,\u00a0B., Shin,\u00a0W., Kim,\u00a0I., Kim,\u00a0S., 2021. Spatially consistent representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1144\u20131153.","DOI":"10.1109\/CVPR46437.2021.00120"},{"key":"10.1016\/j.cviu.2026.104783_b81","doi-asserted-by":"crossref","unstructured":"Selvaraju,\u00a0R.R., Desai,\u00a0K., Johnson,\u00a0J., Naik,\u00a0N., 2021. Casting your model: Learning to localize improves self-supervised representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 11058\u201311067.","DOI":"10.1109\/CVPR46437.2021.01091"},{"key":"10.1016\/j.cviu.2026.104783_b82","doi-asserted-by":"crossref","unstructured":"Silva,\u00a0T., Pedrini,\u00a0H., Ram\u00edrez,\u00a0A., 2023. Self-supervised Learning of Contextualized Local Visual Embeddings. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 177\u2013186.","DOI":"10.1109\/ICCVW60793.2023.00025"},{"key":"10.1016\/j.cviu.2026.104783_b83","series-title":"Dinov3","author":"Sim\u00e9oni","year":"2025"},{"key":"10.1016\/j.cviu.2026.104783_b84","first-page":"1","article-title":"Receptive-field and direction induced attention network for infrared dim small target detection with a large-scale dataset IRDST","volume":"61","author":"Sun","year":"2023","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"issue":"1","key":"10.1016\/j.cviu.2026.104783_b85","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1038\/s41597-023-02066-6","article-title":"HIT-UAV: A high-altitude infrared thermal dataset for unmanned aerial vehicle-based object detection","volume":"10","author":"Suo","year":"2023","journal-title":"Sci. Data"},{"key":"10.1016\/j.cviu.2026.104783_b86","first-page":"11839","article-title":"Csi: Novelty detection via contrastive learning on distributionally shifted instances","volume":"33","author":"Tack","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b87","doi-asserted-by":"crossref","unstructured":"Tao,\u00a0C., Zhu,\u00a0X., Su,\u00a0W., Huang,\u00a0G., Li,\u00a0B., Zhou,\u00a0J., Qiao,\u00a0Y., Wang,\u00a0X., Dai,\u00a0J., 2023. Siamese image modeling for self-supervised vision representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2132\u20132141.","DOI":"10.1109\/CVPR52729.2023.00212"},{"key":"10.1016\/j.cviu.2026.104783_b88","unstructured":"Tian,\u00a0K., Jiang,\u00a0Y., Lin,\u00a0C., Wang,\u00a0L., Yuan,\u00a0Z., et al., 2022. Designing BERT for Convolutional Networks: Sparse and Hierarchical Masked Modeling. In: The Eleventh International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b89","unstructured":"Tomasev,\u00a0N., Bica,\u00a0I., McWilliams,\u00a0B., Buesing,\u00a0L.H., Pascanu,\u00a0R., Blundell,\u00a0C., Mitrovic,\u00a0J., 2022. Pushing the limits of self-supervised ResNets: Can we outperform supervised learning without labels on ImageNet?. In: First Workshop on Pre-Training: Perspectives, Pitfalls, and Paths Forward At ICML 2022."},{"key":"10.1016\/j.cviu.2026.104783_b90","doi-asserted-by":"crossref","unstructured":"Van\u00a0Horn,\u00a0G., Mac\u00a0Aodha,\u00a0O., Song,\u00a0Y., Cui,\u00a0Y., Sun,\u00a0C., Shepard,\u00a0A., Adam,\u00a0H., Perona,\u00a0P., Belongie,\u00a0S., 2018. The inaturalist species classification and detection dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 8769\u20138778.","DOI":"10.1109\/CVPR.2018.00914"},{"key":"10.1016\/j.cviu.2026.104783_b91","series-title":"Franca: Nested matryoshka clustering for scalable visual representation learning","author":"Venkataramanan","year":"2025"},{"key":"10.1016\/j.cviu.2026.104783_b92","doi-asserted-by":"crossref","unstructured":"Vincent,\u00a0P., Larochelle,\u00a0H., Bengio,\u00a0Y., Manzagol,\u00a0P.-A., 2008. Extracting and composing robust features with denoising autoencoders. In: Proceedings of the 25th International Conference on Machine Learning. pp. 1096\u20131103.","DOI":"10.1145\/1390156.1390294"},{"key":"10.1016\/j.cviu.2026.104783_b93","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0Z., Li,\u00a0Q., Zhang,\u00a0G., Wan,\u00a0P., Zheng,\u00a0W., Wang,\u00a0N., Gong,\u00a0M., Liu,\u00a0T., 2022. Exploring set similarity for dense self-supervised representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16590\u201316599.","DOI":"10.1109\/CVPR52688.2022.01609"},{"key":"10.1016\/j.cviu.2026.104783_b94","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0F., Liu,\u00a0H., 2021. Understanding the behaviour of contrastive loss. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2495\u20132504.","DOI":"10.1109\/CVPR46437.2021.00252"},{"key":"10.1016\/j.cviu.2026.104783_b95","series-title":"European Conference on Computer Vision","first-page":"499","article-title":"CP 2: Copy-paste contrastive pretraining for semantic segmentation","author":"Wang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b96","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0X., Zhang,\u00a0R., Shen,\u00a0C., Kong,\u00a0T., Li,\u00a0L., 2021. Dense contrastive learning for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3024\u20133033.","DOI":"10.1109\/CVPR46437.2021.00304"},{"key":"10.1016\/j.cviu.2026.104783_b97","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0H., Zhou,\u00a0L., Wang,\u00a0L., 2019. Miss detection vs. false alarm: Adversarial learning for small object segmentation in infrared images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 8509\u20138518.","DOI":"10.1109\/ICCV.2019.00860"},{"key":"10.1016\/j.cviu.2026.104783_b98","doi-asserted-by":"crossref","unstructured":"Wei,\u00a0C., Fan,\u00a0H., Xie,\u00a0S., Wu,\u00a0C.-Y., Yuille,\u00a0A., Feichtenhofer,\u00a0C., 2022. Masked feature prediction for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14668\u201314678.","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"10.1016\/j.cviu.2026.104783_b99","first-page":"22682","article-title":"Aligning pretraining for detection via object-level contrastive learning","volume":"34","author":"Wei","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b100","series-title":"Detectron2","author":"Wu","year":"2019"},{"key":"10.1016\/j.cviu.2026.104783_b101","doi-asserted-by":"crossref","unstructured":"Xiao,\u00a0T., Reed,\u00a0C.J., Wang,\u00a0X., Keutzer,\u00a0K., Darrell,\u00a0T., 2021. Region similarity representation learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10539\u201310548.","DOI":"10.1109\/ICCV48922.2021.01037"},{"key":"10.1016\/j.cviu.2026.104783_b102","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0E., Ding,\u00a0J., Wang,\u00a0W., Zhan,\u00a0X., Xu,\u00a0H., Sun,\u00a0P., Li,\u00a0Z., Luo,\u00a0P., 2021. Detco: Unsupervised contrastive learning for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 8392\u20138401.","DOI":"10.1109\/ICCV48922.2021.00828"},{"key":"10.1016\/j.cviu.2026.104783_b103","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0Z., Geng,\u00a0Z., Hu,\u00a0J., Zhang,\u00a0Z., Hu,\u00a0H., Cao,\u00a0Y., 2023. Revealing the dark secrets of masked image modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14475\u201314485.","DOI":"10.1109\/CVPR52729.2023.01391"},{"key":"10.1016\/j.cviu.2026.104783_b104","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0Z., Lin,\u00a0Y., Zhang,\u00a0Z., Cao,\u00a0Y., Lin,\u00a0S., Hu,\u00a0H., 2021. Propagate yourself: Exploring pixel-level consistency for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16684\u201316693.","DOI":"10.1109\/CVPR46437.2021.01641"},{"key":"10.1016\/j.cviu.2026.104783_b105","first-page":"28864","article-title":"Unsupervised object-level representation learning from scene images","volume":"34","author":"Xie","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104783_b106","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0Z., Zhang,\u00a0Z., Cao,\u00a0Y., Lin,\u00a0Y., Bao,\u00a0J., Yao,\u00a0Z., Dai,\u00a0Q., Hu,\u00a0H., 2022. Simmim: A simple framework for masked image modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 9653\u20139663.","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"10.1016\/j.cviu.2026.104783_b107","doi-asserted-by":"crossref","unstructured":"Xu,\u00a0J., Lin,\u00a0Z., Zhou,\u00a0D., Yang,\u00a0Y., Liao,\u00a0X., Wang,\u00a0Q., Wu,\u00a0B., Chen,\u00a0G., Heng,\u00a0P.-A., 2024. DPPMask: Masked Image Modeling with Determinantal Point Processes. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2266\u20132276.","DOI":"10.1109\/WACV57701.2024.00226"},{"key":"10.1016\/j.cviu.2026.104783_b108","doi-asserted-by":"crossref","unstructured":"Xue,\u00a0H., Gao,\u00a0P., Li,\u00a0H., Qiao,\u00a0Y., Sun,\u00a0H., Li,\u00a0H., Luo,\u00a0J., 2023. Stare at what you see: Masked image modeling without reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 22732\u201322741.","DOI":"10.1109\/CVPR52729.2023.02177"},{"key":"10.1016\/j.cviu.2026.104783_b109","doi-asserted-by":"crossref","unstructured":"Yang,\u00a0C., Wu,\u00a0Z., Zhou,\u00a0B., Lin,\u00a0S., 2021. Instance localization for self-supervised detection pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3987\u20133996.","DOI":"10.1109\/CVPR46437.2021.00398"},{"key":"10.1016\/j.cviu.2026.104783_b110","series-title":"Inscon: Instance consistency feature representation via self-supervised learning","author":"Yang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104783_b111","doi-asserted-by":"crossref","unstructured":"Yun,\u00a0S., Han,\u00a0D., Oh,\u00a0S.J., Chun,\u00a0S., Choe,\u00a0J., Yoo,\u00a0Y., 2019. Cutmix: Regularization strategy to train strong classifiers with localizable features. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6023\u20136032.","DOI":"10.1109\/ICCV.2019.00612"},{"key":"10.1016\/j.cviu.2026.104783_b112","series-title":"International Conference on Machine Learning","first-page":"12310","article-title":"Barlow twins: Self-supervised learning via redundancy reduction","author":"Zbontar","year":"2021"},{"key":"10.1016\/j.cviu.2026.104783_b113","series-title":"Mixup: Beyond empirical risk minimization","author":"Zhang","year":"2017"},{"key":"10.1016\/j.cviu.2026.104783_b114","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0C., Zhang,\u00a0K., Pham,\u00a0T.X., Niu,\u00a0A., Qiao,\u00a0Z., Yoo,\u00a0C.D., Kweon,\u00a0I.S., 2022. Dual temperature helps contrastive learning without many negative samples: Towards understanding and simplifying moco. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14441\u201314450.","DOI":"10.1109\/CVPR52688.2022.01404"},{"key":"10.1016\/j.cviu.2026.104783_b115","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0M., Zhang,\u00a0R., Yang,\u00a0Y., Bai,\u00a0H., Zhang,\u00a0J., Guo,\u00a0J., 2022. ISNet: Shape Matters for Infrared Small Target Detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 877\u2013886.","DOI":"10.1109\/CVPR52688.2022.00095"},{"key":"10.1016\/j.cviu.2026.104783_b116","doi-asserted-by":"crossref","unstructured":"Zhao,\u00a0Y., Wang,\u00a0G., Luo,\u00a0C., Zeng,\u00a0W., Zha,\u00a0Z.-J., 2021. Self-supervised visual representations learning by contrastive mask prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10160\u201310169.","DOI":"10.1109\/ICCV48922.2021.01000"},{"key":"10.1016\/j.cviu.2026.104783_b117","unstructured":"Zhou,\u00a0J., Wei,\u00a0C., Wang,\u00a0H., Shen,\u00a0W., Xie,\u00a0C., Yuille,\u00a0A., Kong,\u00a0T., 2021. Image BERT Pre-training with Online Tokenizer. In: International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104783_b118","doi-asserted-by":"crossref","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","article-title":"Semantic understanding of scenes through the ade20k dataset","volume":"127","author":"Zhou","year":"2019","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.cviu.2026.104783_b119","doi-asserted-by":"crossref","unstructured":"Ziegler,\u00a0A., Asano,\u00a0Y.M., 2022. Self-supervised learning of object parts for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14502\u201314511.","DOI":"10.1109\/CVPR52688.2022.01410"},{"key":"10.1016\/j.cviu.2026.104783_b120","series-title":"European Conference on Computer Vision","first-page":"392","article-title":"Spot-the-difference self-supervised pre-training for anomaly detection and segmentation","author":"Zou","year":"2022"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001505?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001505?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T22:33:19Z","timestamp":1778797999000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001505"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":120,"alternative-id":["S1077314226001505"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104783","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Self-supervised learning for object detection in challenging settings: A survey","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104783","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104783"}}