{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T04:10:32Z","timestamp":1784607032494,"version":"3.55.0"},"reference-count":270,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172090"],"award-info":[{"award-number":["62172090"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23B2054"],"award-info":[{"award-number":["U23B2054"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276263"],"award-info":[{"award-number":["62276263"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Start-up Research Fund of Southeast University","award":["RF1028623097"],"award-info":[{"award-number":["RF1028623097"]}]},{"name":"NTU RSR"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tpami.2024.3415112","type":"journal-article","created":{"date-parts":[[2024,6,17]],"date-time":"2024-06-17T18:12:40Z","timestamp":1718647960000},"page":"9052-9071","source":"Crossref","is-referenced-by-count":481,"title":["A Survey on Self-Supervised Learning: Algorithms, Applications, and Future Trends"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9450-1759","authenticated-orcid":false,"given":"Jie","family":"Gui","sequence":"first","affiliation":[{"name":"School of Cyber Science and Engineering, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4837-2003","authenticated-orcid":false,"given":"Tuo","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Cyber Science and Engineering, Southeast University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6595-7661","authenticated-orcid":false,"given":"Jing","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Camperdown, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8750-3505","authenticated-orcid":false,"given":"Qiong","family":"Cao","sequence":"additional","affiliation":[{"name":"JD Explore Academy, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4029-9935","authenticated-orcid":false,"given":"Zhenan","family":"Sun","sequence":"additional","affiliation":[{"name":"Center for Research on Intelligent Perception and Computing, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6405-4011","authenticated-orcid":false,"given":"Hao","family":"Luo","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7225-5449","authenticated-orcid":false,"given":"Dacheng","family":"Tao","sequence":"additional","affiliation":[{"name":"College of Computing &amp; Data Science at Nanyang Technological University, Nanyang Avenue, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref2","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00537"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3090866"},{"key":"ref5","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref7","first-page":"1","article-title":"Unsupervised representation learning by predicting image rotations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gidaris"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_5"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_32"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00840"},{"key":"ref11","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6970"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"ref14","article-title":"A critical analysis of self-supervision, or what we can learn from a single image","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Asano"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1126\/science.1127647"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390294"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487517"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.179"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_41"},{"key":"ref20","article-title":"Rethinking data augmentation: Self-supervision and self-distillation","author":"Lee","year":"2019"},{"key":"ref21","first-page":"3833","article-title":"Rethinking pre-training and self-training","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zoph"},{"key":"ref22","first-page":"9960","article-title":"Self-supervised learning through the eyes of a child","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Orhan"},{"key":"ref23","first-page":"1","article-title":"Representation learning via invariant causal mechanisms","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Mitrovic"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00946"},{"key":"ref25","article-title":"Yann LeCun, Yoshua Bengio: Self-supervised learning is key to human-level intelligence","year":"2020"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2022.109126"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3172903"},{"key":"ref28","article-title":"A survey on self-supervised pre-training for sequential transfer learning in neural networks","author":"Mao","year":"2020"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3577925"},{"key":"ref30","article-title":"Adversarial pretraining of self-supervised deep networks: Past, present and future","author":"Qi","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.3390\/technologies9010002"},{"key":"ref32","first-page":"112","article-title":"Learning classification with unlabeled data","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"de Sa"},{"key":"ref33","article-title":"Reflections from the turing award winners","author":"LeCun","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992393"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2021.3130191"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00973"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.13"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_40"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_35"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073703"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.96"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00649"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2019.00025"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00198"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2892452"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.628"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00393"},{"key":"ref49","first-page":"1","article-title":"What makes instance discrimination good for transfer learning?","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhao"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref51","article-title":"Improved baselines with momentum contrastive learning","author":"Chen","year":"2020"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref53","first-page":"22243","article-title":"Big self-supervised models are strong semi-supervised learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref54","first-page":"9929","article-title":"Understanding contrastive representation learning through alignment and uniformity on the hypersphere","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang"},{"key":"ref55","first-page":"12310","article-title":"Barlow twins: Self-supervised learning via redundancy reduction","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zbontar"},{"key":"ref56","first-page":"1","article-title":"VICReg: Variance-invariance-covariance regularization for self-supervised learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bardes"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"ref58","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2019"},{"key":"ref59","first-page":"297","article-title":"Noise-contrastive estimation: A new estimation principle for unnormalized statistical models","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Gutmann"},{"key":"ref60","article-title":"ReSSL: Relational self-supervised learning with weak augmentation","author":"Zheng","year":"2021"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17312"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"ref64","first-page":"6827","article-title":"What makes for good views for contrastive learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Tian"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01641"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3203630"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497510"},{"key":"ref68","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Caron"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref71","first-page":"1","article-title":"On mutual information maximization for representation learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Tschannen"},{"key":"ref72","first-page":"5628","article-title":"A theoretical analysis of contrastive unsupervised representation learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Saunshi"},{"key":"ref73","first-page":"19290","article-title":"Rethinking the value of labels for improving class-imbalanced learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref74","article-title":"Self-supervised learning from a multi-view perspective","author":"Tsai","year":"2020"},{"key":"ref75","article-title":"Debiased contrastive learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chuang"},{"key":"ref76","article-title":"Predicting what you already know helps: Provable self-supervised learning","author":"Lee","year":"2020"},{"key":"ref77","first-page":"1673","article-title":"Large-margin contrastive learning with distance polarization regularizer","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chen"},{"key":"ref78","first-page":"5000","article-title":"Provable guarantees for self-supervised deep learning with spectral contrastive loss","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"HaoChen"},{"key":"ref79","first-page":"1179","article-title":"Contrastive learning, multi-view redundancy, and linear models","volume-title":"Proc. Int. Conf. Algorithmic Learn. Theory","author":"Tosh"},{"key":"ref80","first-page":"1","article-title":"Theoretical analysis of self-training with deep networks on unlabeled data","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wei"},{"key":"ref81","article-title":"Deep contrastive learning is provably (almost) principal component analysis","author":"Tian","year":"2022"},{"key":"ref82","first-page":"9640","article-title":"An empirical study of training self-supervised visual transformers","volume-title":"Proc. IEEE Int. Conf. Comput. Vis.","author":"Chen"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01414"},{"key":"ref85","article-title":"Deep unsupervised learning through spatial contrasting","author":"Hoffer","year":"2016"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_28"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547784"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01014"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00119"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00872"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00935"},{"key":"ref92","first-page":"1","article-title":"Understanding dimensional collapse in contrastive self-supervised learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jing"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475510"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00113"},{"key":"ref95","first-page":"21798","article-title":"Hard negative mixing for contrastive learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kalantidis"},{"key":"ref96","first-page":"3407","article-title":"Demystifying contrastive self-supervised learning: Invariances, augmentations and dataset biases","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Purushwalkam"},{"key":"ref97","first-page":"18661","article-title":"Supervised contrastive learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Khosla"},{"key":"ref98","first-page":"1","article-title":"iBOT: Image bert pre-training with online tokenizer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhou"},{"key":"ref99","first-page":"1","article-title":"BEiT: Bert pre-training of image transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bao"},{"key":"ref100","article-title":"Context autoencoder for self-supervised representation learning","author":"Chen","year":"2022"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"ref102","article-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"ref103","first-page":"1691","article-title":"Generative pretraining from pixels","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chen"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.278"},{"key":"ref105","first-page":"8821","article-title":"Zero-shot text-to-image generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ramesh"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25130"},{"key":"ref108","article-title":"Data2vec: A general framework for self-supervised learning in speech, vision and language","author":"Baevski","year":"2022"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20056-4_7"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548173"},{"key":"ref111","article-title":"BEiT v2: Masked image modeling with vector-quantized visual tokenizers","author":"Peng","year":"2022"},{"key":"ref112","article-title":"Masked autoencoders as spatiotemporal learners","author":"Feichtenhofer","year":"2022"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20062-5_3"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20086-1_35"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01432"},{"key":"ref116","first-page":"10078","article-title":"VideoMAE: Masked autoencoders are data-efficient learners for self-supervised video pre-training","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Tong"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01003"},{"key":"ref118","first-page":"40676","article-title":"Siamese masked autoencoders","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Gupta"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01170"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_17"},{"key":"ref121","first-page":"38571","article-title":"ViTPose: Simple vision transformer baselines for human pose estimation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Xu"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25269"},{"key":"ref123","article-title":"Contrast with reconstruct: Contrastive 3D representation learning guided by generative pretraining","author":"Qi","year":"2023"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00999"},{"key":"ref125","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2023"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00604"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00204"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/200"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3336525"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00212"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01391"},{"key":"ref132","first-page":"766","article-title":"Discriminative unsupervised feature learning with convolutional neural networks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dosovitskiy"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2015.2496141"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"ref135","first-page":"517","article-title":"Unsupervised learning by predicting noise","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bojanowski"},{"key":"ref136","first-page":"478","article-title":"Unsupervised deep embedding for clustering analysis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xie"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.556"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.76"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.149"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00202"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00312"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969125"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01243"},{"key":"ref145","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00156"},{"key":"ref146","first-page":"15663","article-title":"Using self-supervised learning can improve model robustness and uncertainty","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hendrycks"},{"key":"ref147","first-page":"4116","article-title":"Contrastive multi-view representation learning on graphs","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hassani"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.218"},{"key":"ref149","article-title":"Self-supervised feature learning by cross-modality and cross-view correspondences","author":"Jing","year":"2020"},{"key":"ref150","article-title":"Self-supervised modal and view invariant feature learning","author":"Jing","year":"2020"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2019.00051"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00029"},{"key":"ref153","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_7"},{"key":"ref154","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00110"},{"key":"ref155","first-page":"9229","article-title":"Test-time training with self-supervision for generalization under distribution shifts","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sun"},{"key":"ref156","article-title":"Test-time training with masked autoencoders","author":"Gandelsman","year":"2022"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00290"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00086"},{"key":"ref159","first-page":"16282","article-title":"Universal domain adaptation through self supervision","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Saito"},{"key":"ref160","article-title":"Unsupervised domain adaptation through self-supervision","author":"Sun","year":"2019"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00975"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403237"},{"key":"ref163","first-page":"12559","article-title":"Self-supervised graph transformer on large-scale molecular data","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Rong"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_47"},{"key":"ref165","article-title":"Bootstrap latent-predictive representations for multitask reinforcement learning","author":"Guo","year":"2020"},{"key":"ref166","article-title":"Self-supervised policy adaptation during deployment","author":"Hansen","year":"2020"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00815"},{"key":"ref168","article-title":"Boosting supervision with self-supervision for few-shot learning","author":"Su","year":"2019"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01206"},{"key":"ref170","first-page":"21480","article-title":"When does contrastive learning preserve adversarial robustness from pretraining to finetuning?","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Fan"},{"key":"ref171","first-page":"2983","article-title":"Adversarial self-supervised contrastive learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kim"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00078"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00813"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/295"},{"key":"ref175","first-page":"117","article-title":"Computer recognition of vowel sounds using a self-supervised learning algorithm","volume":"6","author":"Pal","year":"1978","journal-title":"J. Anat. Soc. India"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.1109\/TFUZZ.1993.390285"},{"key":"ref177","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-49409-8_20"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.715"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2820063"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00384"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.638"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01229"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475287"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01469"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00841"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00637"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_6"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_34"},{"key":"ref189","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.281"},{"key":"ref190","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01003"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2947204"},{"key":"ref192","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_18"},{"key":"ref193","first-page":"20320","article-title":"Noise2Same: Optimizing a self-supervised bound for image denoising","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Xie"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01454"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00398"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.465"},{"key":"ref197","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00828"},{"key":"ref198","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2023.3290038"},{"key":"ref199","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00251"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_29"},{"key":"ref201","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.166"},{"key":"ref202","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00212"},{"key":"ref203","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01367"},{"key":"ref204","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00784"},{"key":"ref205","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01382"},{"key":"ref206","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3532010"},{"key":"ref207","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00186"},{"key":"ref208","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_19"},{"key":"ref209","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.607"},{"key":"ref210","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.79"},{"key":"ref211","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01058"},{"key":"ref212","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00994"},{"key":"ref213","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00658"},{"key":"ref214","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58520-4_30"},{"key":"ref215","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00629"},{"key":"ref216","first-page":"5679","article-title":"Self-supervised co-training for video representation learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Han"},{"key":"ref217","first-page":"7763","article-title":"Cooperative learning of audio and video models from self-supervised synchronization","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Korbar"},{"key":"ref218","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"ref219","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref220","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01033"},{"key":"ref221","article-title":"Learning video representations from textual web supervision","author":"Stroud","year":"2020"},{"key":"ref222","article-title":"Self-supervised multimodal versatile networks","author":"Alayrac","year":"2020"},{"key":"ref223","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8462891"},{"key":"ref224","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00267"},{"key":"ref225","first-page":"318","article-title":"Joint-task self-supervised learning for temporal correspondence","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref226","first-page":"19545","article-title":"Space-time correspondence as a contrastive random walk","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Jabri"},{"key":"ref227","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00651"},{"key":"ref228","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00455"},{"key":"ref229","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6840"},{"key":"ref230","first-page":"4182","article-title":"Data-efficient image recognition with contrastive predictive coding","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"H\u00e9naff"},{"key":"ref231","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref232","article-title":"Efficient self-supervised vision transformers for representation learning","author":"Li","year":"2021"},{"key":"ref233","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1310.4546"},{"key":"ref234","article-title":"ELECTRA: Pre-training text encoders as discriminators rather than generators","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Clark"},{"key":"ref235","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00259"},{"key":"ref236","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.20"},{"key":"ref237","article-title":"CLEAR: Contrastive learning for sentence representation","author":"Wu","year":"2020"},{"key":"ref238","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.72"},{"key":"ref239","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00348"},{"key":"ref240","first-page":"12546","article-title":"Contrastive learning of global and local features for medical image segmentation with limited annotations","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Chaitanya"},{"key":"ref241","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2020.101746"},{"key":"ref242","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00928"},{"key":"ref243","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3222818"},{"key":"ref244","article-title":"MixMIM: Mixed and masked image modeling for efficient visual representation learning","author":"Liu","year":"2022"},{"key":"ref245","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.354"},{"key":"ref246","first-page":"10929","article-title":"RankMe: Assessing the downstream performance of pretrained self-supervised representations by their rank","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Garrido"},{"key":"ref247","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"ref248","first-page":"740","article-title":"Microsoft COCO: Common objects in context","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Lin"},{"key":"ref249","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.544"},{"key":"ref250","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1140-0"},{"key":"ref251","article-title":"The kinetics human action video dataset","author":"Kay","year":"2017"},{"key":"ref252","first-page":"5842","article-title":"The \u201csomething something","volume-title":"Proc. IEEE Int. Conf. Comput. Vis.","author":"Goyal"},{"key":"ref253","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00633"},{"key":"ref254","article-title":"UCF101: A dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"ref255","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"ref256","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i11.17215"},{"key":"ref257","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412071"},{"key":"ref258","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00129"},{"key":"ref259","article-title":"Video representation learning with visual tempo consistency","author":"Yang","year":"2020"},{"key":"ref260","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00331"},{"key":"ref261","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"ref262","first-page":"4974","article-title":"Can contrastive learning avoid shortcut solutions?","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Robinson"},{"key":"ref263","article-title":"Contrastive learning rivals masked image modeling in fine-tuning via feature distillation","author":"Wei","year":"2022"},{"key":"ref264","first-page":"11834","article-title":"Intriguing properties of contrastive losses","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst","author":"Chen"},{"key":"ref265","first-page":"10268","article-title":"Understanding self-supervised learning dynamics without contrastive pairs","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tian"},{"key":"ref266","article-title":"On the duality between contrastive and non-contrastive self-supervised learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Garrido"},{"key":"ref267","article-title":"Simplicial embeddings in self-supervised learning and downstream classification","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lavoie"},{"key":"ref268","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01403"},{"key":"ref269","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00304"},{"key":"ref270","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01838"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10746266\/10559458.pdf?arnumber=10559458","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:27:07Z","timestamp":1732667227000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10559458\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":270,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3415112","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}