{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T19:38:46Z","timestamp":1785440326009,"version":"3.56.0"},"reference-count":218,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T00:00:00Z","timestamp":1659657600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T00:00:00Z","timestamp":1659657600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s13735-022-00245-6","type":"journal-article","created":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T12:17:04Z","timestamp":1659701824000},"page":"461-488","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":103,"title":["Contrastive self-supervised learning: review, progress, challenges and future research directions"],"prefix":"10.1007","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6167-9884","authenticated-orcid":false,"given":"Pranjal","family":"Kumar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Piyush","family":"Rawat","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siddhartha","family":"Chauhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,8,5]]},"reference":[{"key":"245_CR1","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition. IEEE, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"245_CR2","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"245_CR3","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"245_CR4","doi-asserted-by":"crossref","unstructured":"Girshick R, Donahue J, Darrell T, Malik J (2014) Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 580\u2013587","DOI":"10.1109\/CVPR.2014.81"},{"key":"245_CR5","doi-asserted-by":"crossref","unstructured":"Long J, Shelhamer E, Darrell T (2015) Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3431\u20133440","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"1","key":"245_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-031-02145-9","volume":"5","author":"B Liu","year":"2012","unstructured":"Liu B (2012) Sentiment analysis and opinion mining. Synth Lect Hum Lang Technol 5(1):1\u2013167","journal-title":"Synth Lect Hum Lang Technol"},{"key":"245_CR7","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805"},{"key":"245_CR8","unstructured":"Lan Z, Chen M, Goodman S, Gimpel K, Sharma P, Soricut R (2019) Albert: A lite Bert for self-supervised learning of language representations. arXiv:1909.11942"},{"key":"245_CR9","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) Roberta: a robustly optimized Bert pretraining approach. arXiv:1907.11692"},{"key":"245_CR10","unstructured":"Yang Z, Dai Z, Yang Y, Carbonell J, Salakhutdinov RR, Le QV (2019) Xlnet: Generalized autoregressive pretraining for language understanding. Adv Neural Inf Process Syst 32"},{"key":"245_CR11","unstructured":"Asai A, Hashimoto K, Hajishirzi H, Socher R, Xiong C (2019) Learning to retrieve reasoning paths over wikipedia graph for question answering. arXiv:1911.10470"},{"key":"245_CR12","doi-asserted-by":"crossref","unstructured":"Ding M, Zhou C, Chen Q, Yang H, Tang J (2019) Cognitive graph for multi-hop reading comprehension at scale. arXiv:1905.05460","DOI":"10.18653\/v1\/P19-1259"},{"key":"245_CR13","doi-asserted-by":"crossref","unstructured":"Rajpurkar P, Zhang J, Lopyrev K, Liang P (2016) Squad: 100,000+ questions for machine comprehension of text. arXiv:1606.05250","DOI":"10.18653\/v1\/D16-1264"},{"key":"245_CR14","doi-asserted-by":"crossref","unstructured":"Yang Z, Qi P, Zhang S, Bengio Y, Cohen WW, Salakhutdinov R, Manning CD (2018) Hotpotqa: a dataset for diverse, explainable multi-hop question answering. arXiv:1809.09600","DOI":"10.18653\/v1\/D18-1259"},{"key":"245_CR15","doi-asserted-by":"crossref","unstructured":"Selvaraju RR, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-cam: visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE international conference on computer vision, pp 618\u2013626","DOI":"10.1109\/ICCV.2017.74"},{"key":"245_CR16","unstructured":"Kalantidis Y, Sariyildiz M, Weinzaepfel P, Larlus D (2020) Improving self-supervised representation learning by synthesizing challenging negatives. Naver Labs Europe"},{"issue":"8","key":"245_CR17","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio Y, Courville A, Vincent P (2013) Representation learning: a review and new perspectives. IEEE Trans Pattern Anal Mach Intell 35(8):1798\u20131828","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"245_CR18","unstructured":"Zimmermann RS, Sharma Y, Schneider S, Bethge M, Brendel W (2021) Contrastive learning inverts the data generating process. In: International conference on machine learning. PMLR, pp 12979\u201312990"},{"key":"245_CR19","doi-asserted-by":"crossref","unstructured":"Ili\u0107 S, Marrese-Taylor E, Balazs JA, Matsuo Y (2018) Deep contextualized word representations for detecting sarcasm and irony. arXiv:1809.09795","DOI":"10.18653\/v1\/W18-6202"},{"key":"245_CR20","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I (2018) Improving language understanding by generative pre-training"},{"key":"245_CR21","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, Subbiah M, Kaplan JD, Dhariwal P, Neelakantan A, Shyam P, Sastry G, Askell A et al (2020) Language models are few-shot learners. Adv Neural Inf Process Syst 33:1877\u20131901","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR22","unstructured":"Van den Oord A, Li Y, Vinyals O (2018) Representation learning with contrastive predictive coding. arXiv:1807.03748"},{"key":"245_CR23","doi-asserted-by":"crossref","unstructured":"Schneider S, Baevski A, Collobert R, Auli M (2019) wav2vec: Unsupervised pre-training for speech recognition. arXiv:1904.05862","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"245_CR24","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) wav2vec 2.0: A framework for self-supervised learning of speech representations. Adv Neural Inf Process Syst 33:12449\u201312460","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR25","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton G (2020) A simple framework for contrastive learning of visual representations. In: International conference on machine learning. PMLR, pp 1597\u20131607"},{"key":"245_CR26","doi-asserted-by":"crossref","unstructured":"Chen X, Xie S, He K (2021) An empirical study of training self-supervised vision transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 9640\u20139649","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"245_CR27","doi-asserted-by":"crossref","unstructured":"Caron M, Touvron H, Misra I, J\u00e9gou H, Mairal J, Bojanowski P, Joulin A (2021) Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 9650\u20139660","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"245_CR28","unstructured":"Bao H, Dong L, Wei F (2021) Beit: Bert pre-training of image transformers. arXiv:2106.08254"},{"key":"245_CR29","doi-asserted-by":"crossref","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick R (2021) Masked autoencoders are scalable vision learners. arXiv:2111.06377","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"245_CR30","unstructured":"Lample G, Conneau A, Denoyer L, Ranzato M (2017) Unsupervised machine translation using monolingual corpora only. arXiv:1711.00043"},{"key":"245_CR31","unstructured":"Baevski A, Hsu W-N, Conneau A, Auli M (2021) Unsupervised speech recognition. Adv Neural Inf Process Syst 34"},{"key":"245_CR32","doi-asserted-by":"crossref","unstructured":"Hsu W-N, Tsai Y-HH, Bolte B, Salakhutdinov R, Mohamed A (2021) Hubert: how much can a bad teacher benefit ASR pre-training?. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6533\u20136537","DOI":"10.1109\/ICASSP39728.2021.9414460"},{"key":"245_CR33","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning. PMLR, pp 8748\u20138763"},{"key":"245_CR34","first-page":"21271","volume":"33","author":"J-B Grill","year":"2020","unstructured":"Grill J-B, Strub F, Altch\u00e9 F, Tallec C, Richemond P, Buchatskaya E, Doersch C, Avila Pires B, Guo Z, Gheshlaghi Azar M et al (2020) Bootstrap your own latent-a new approach to self-supervised learning. Adv Neural Inf Process Syst 33:21271\u201321284","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1521","key":"245_CR35","doi-asserted-by":"publisher","first-page":"1211","DOI":"10.1098\/rstb.2008.0300","volume":"364","author":"K Friston","year":"2009","unstructured":"Friston K, Kiebel S (2009) Predictive coding under the free-energy principle. Philos Trans R Soc B Bioll Sci 364(1521):1211\u20131221","journal-title":"Philos Trans R Soc B Bioll Sci"},{"issue":"2","key":"245_CR36","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1038\/nrn2787","volume":"11","author":"K Friston","year":"2010","unstructured":"Friston K (2010) The free-energy principle: A unified brain theory? Nat Rev Neurosci 11(2):127\u2013138","journal-title":"Nat Rev Neurosci"},{"key":"245_CR37","unstructured":"Jaegle A, Gimeno F, Brock A, Vinyals O, Zisserman A, Carreira J (2021) Perceiver: general perception with iterative attention. In: International conference on machine learning. PMLR, pp 4651\u20134664"},{"issue":"11","key":"245_CR38","doi-asserted-by":"publisher","first-page":"719","DOI":"10.1038\/s42256-020-00247-1","volume":"2","author":"OG Holmberg","year":"2020","unstructured":"Holmberg OG, K\u00f6hler ND, Martins T, Siedlecki J, Herold T, Keidel L, Asani B, Schiefelbein J, Priglinger S, Kortuem KU et al (2020) Self-supervised retinal thickness prediction enables deep learning from unlabelled data to boost classification of diabetic retinopathy. Nat Mach Intell 2(11):719\u2013726","journal-title":"Nat Mach Intell"},{"key":"245_CR39","doi-asserted-by":"crossref","unstructured":"Arandjelovic R, Zisserman A (2017) Look, listen and learn. In: Proceedings of the IEEE international conference on computer vision, pp 609\u2013617","DOI":"10.1109\/ICCV.2017.73"},{"key":"245_CR40","doi-asserted-by":"crossref","unstructured":"Arandjelovic R, Zisserman A (2018) Objects that sound. In: Proceedings of the European conference on computer vision (ECCV), pp 435\u2013451","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"245_CR41","doi-asserted-by":"crossref","unstructured":"Lee H-Y, Huang J-B, Singh M, Yang M-H (2017) Unsupervised representation learning by sorting sequences. In: Proceedings of the IEEE international conference on computer vision, pp 667\u2013676","DOI":"10.1109\/ICCV.2017.79"},{"key":"245_CR42","doi-asserted-by":"crossref","unstructured":"Misra I, van der Maaten L (2020) Self-supervised learning of pretext-invariant representations. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6707\u20136717","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"245_CR43","doi-asserted-by":"crossref","unstructured":"Fernando B, Bilen H, Gavves E, Gould S (2017) Self-supervised video representation learning with odd-one-out networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3636\u20133645","DOI":"10.1109\/CVPR.2017.607"},{"key":"245_CR44","doi-asserted-by":"crossref","unstructured":"Wei D, Lim JJ, Zisserman A, Freeman WT (2018) Learning and using the arrow of time. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 8052\u20138060","DOI":"10.1109\/CVPR.2018.00840"},{"key":"245_CR45","doi-asserted-by":"crossref","unstructured":"Gan C, Gong B, Liu K, Su H, Guibas LJ (2018) Geometry guided convolutional neural networks for self-supervised video representation learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5589\u20135597","DOI":"10.1109\/CVPR.2018.00586"},{"key":"245_CR46","unstructured":"Vondrick C, Pirsiavash H, Torralba A (2016) Generating videos with scene dynamics. Adv Neural Inf Process Syst 29"},{"key":"245_CR47","doi-asserted-by":"crossref","unstructured":"Zhao Y, Deng B, Shen C, Liu Y, Lu H, Hua X-S (2017) Spatio-temporal autoencoder for video anomaly detection. In: Proceedings of the 25th ACM international conference on multimedia, pp 1933\u20131941","DOI":"10.1145\/3123266.3123451"},{"issue":"01","key":"245_CR48","first-page":"8545","volume":"33","author":"D Kim","year":"2019","unstructured":"Kim D, Cho D, Kweon IS (2019) Self-supervised video representation learning with space-time cubic puzzles. Proc AAAI Conf Artif Intell 33(01):8545\u20138552","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"245_CR49","first-page":"5679","volume":"33","author":"T Han","year":"2020","unstructured":"Han T, Xie W, Zisserman A (2020) Self-supervised co-training for video representation learning. Adv Neural Inf Process Syst 33:5679\u20135690","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR50","first-page":"8089","volume":"33","author":"Q Kong","year":"2020","unstructured":"Kong Q, Wei W, Deng Z, Yoshinaga T, Murakami T (2020) Cycle-contrast for self-supervised video representation learning. Adv Neural Inf Process Syst 33:8089\u20138100","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR51","doi-asserted-by":"crossref","unstructured":"Qian R, Meng T, Gong B, Yang M-H, Wang H, Belongie S, Cui Y (2021) Spatiotemporal contrastive video representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6964\u20136974","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"245_CR52","unstructured":"McCann B, Bradbury J, Xiong C, Socher R (2017) Learned in translation: contextualized word vectors. Adv Neural Inf Process Syst 30"},{"key":"245_CR53","doi-asserted-by":"crossref","unstructured":"Baevski A, Edunov S, Liu Y, Zettlemoyer L, Auli M (2019) Cloze-driven pretraining of self-attention networks. arXiv:1903.07785","DOI":"10.18653\/v1\/D19-1539"},{"key":"245_CR54","doi-asserted-by":"crossref","unstructured":"Jiao X, Yin Y, Shang L, Jiang X, Chen X, Li L, Wang F, Liu Q (2019) Tinybert: distilling Bert for natural language understanding. arXiv:1909.10351","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"245_CR55","doi-asserted-by":"crossref","unstructured":"Baevski A, Auli M, Mohamed A (2019) Effectiveness of self-supervised pre-training for speech recognition. arXiv:1911.03912","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"key":"245_CR56","unstructured":"Baevski A, Schneider S, Auli M (2019) vq-wav2vec: Self-supervised learning of discrete speech representations. arXiv:1910.05453"},{"key":"245_CR57","unstructured":"Zhang Y, Qin J, Park DS, Han W, Chiu C-C, Pang R, Le QV, Wu Y (2020) Pushing the limits of semi-supervised learning for automatic speech recognition. arXiv:2010.10504"},{"key":"245_CR58","doi-asserted-by":"crossref","unstructured":"Chung Y-A, Zhang Y, Han W, Chiu C-C, Qin J, Pang R, Wu Y (2021) W2v-bert: Combining contrastive learning and masked language modeling for self-supervised speech pre-training. arXiv:2108.06209","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"245_CR59","doi-asserted-by":"crossref","unstructured":"Zhang Y, Park DS, Han W, Qin J, Gulati A, Shor J, Jansen A, Xu Y, Huang Y, Wang S et\u00a0al (2021) Bigssl: exploring the frontier of large-scale semi-supervised learning for automatic speech recognition. arXiv:2109.13226","DOI":"10.1109\/JSTSP.2022.3182537"},{"key":"245_CR60","unstructured":"Chiu C-C, Qin J, Zhang Y, Yu J, Wu Y (2022) Self-supervised learning with random-projection quantizer for speech recognition. arXiv:2202.01855"},{"key":"245_CR61","doi-asserted-by":"crossref","unstructured":"Liu X, Zhang F, Hou Z, Mian L, Wang Z, Zhang J, Tang J (2021) Self-supervised learning: Generative or contrastive. IEEE Trans Knowl Data Eng","DOI":"10.1109\/TKDE.2021.3090866"},{"issue":"8","key":"245_CR62","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford A, Wu J, Child R, Luan D, Amodei D, Sutskever I et al (2019) Language models are unsupervised multitask learners. OpenAI Blog 1(8):9","journal-title":"OpenAI Blog"},{"key":"245_CR63","unstructured":"Tran C, Bhosale S, Cross J, Koehn P, Edunov S, Fan A (2021) Facebook ai wmt21 news translation task submission. arXiv:2108.03265"},{"key":"245_CR64","unstructured":"Arivazhagan N, Bapna A, Firat O, Lepikhin D, Johnson M, Krikun M, Chen MX, Cao Y, Foster G, Cherry C et\u00a0al (2019) Massively multilingual neural machine translation in the wild: findings and challenges. arXiv:1907.05019"},{"key":"245_CR65","unstructured":"Van\u00a0Oord A, Kalchbrenner N, Kavukcuoglu K (2016) Pixel recurrent neural networks. In: International conference on machine learning. PMLR, pp 1747\u20131756"},{"key":"245_CR66","unstructured":"Van\u00a0den Oord A, Kalchbrenner N, Espeholt L, Vinyals O, Graves A et\u00a0al (2016) Conditional image generation with Pixelcnn decoders. Adv Neural Inf Process Syst 29"},{"key":"245_CR67","unstructured":"Rezende D, Mohamed S (2015) Variational inference with normalizing flows. In: International conference on machine learning. PMLR, pp 1530\u20131538"},{"key":"245_CR68","doi-asserted-by":"crossref","unstructured":"Yang G, Huang X, Hao Z, Liu M-Y, Belongie S, Hariharan B (2019) Pointflow: 3d point cloud generation with continuous normalizing flows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4541\u20134550","DOI":"10.1109\/ICCV.2019.00464"},{"key":"245_CR69","first-page":"19667","volume":"33","author":"A Vahdat","year":"2020","unstructured":"Vahdat A, Kautz J (2020) Nvae: a deep hierarchical variational autoencoder. Adv Neural Inf Process Syst 33:19667\u201319679","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR70","unstructured":"Chen M, Radford A, Child R, Wu J, Jun H, Luan D, Sutskever I (2020) Generative pretraining from pixels. In: International conference on machine learning. PMLR, pp 1691\u20131703"},{"key":"245_CR71","unstructured":"You J, Ying R, Ren X, Hamilton W, Leskovec J (2018) Graphrnn: generating realistic graphs with deep auto-regressive models. In: International conference on machine learning. PMLR, pp 5708\u20135717"},{"key":"245_CR72","doi-asserted-by":"publisher","DOI":"10.1016\/j.ress.2021.107805","volume":"215","author":"L Zhang","year":"2021","unstructured":"Zhang L, Lin J, Shao H, Zhang Z, Yan X, Long J (2021) End-to-end unsupervised fault detection using a flow-based model. Reliab Eng Syst Saf 215:107805","journal-title":"Reliab Eng Syst Saf"},{"key":"245_CR73","unstructured":"Hinton GE, Zemel R (1993) Autoencoders, minimum description length and helmholtz free energy. Adv Neural Inf Process Syst 6"},{"issue":"3","key":"245_CR74","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1162\/089976600300015691","volume":"12","author":"N Japkowicz","year":"2000","unstructured":"Japkowicz N, Hanson SJ, Gluck MA (2000) Nonlinear autoassociation is not equivalent to PCA. Neural Comput 12(3):531\u2013545","journal-title":"Neural Comput"},{"key":"245_CR75","doi-asserted-by":"crossref","unstructured":"Vincent P, Larochelle H, Bengio Y, Manzagol P-A (2008) Extracting and composing robust features with denoising autoencoders. In: Proceedings of the 25th international conference on machine learning, pp 1096\u20131103","DOI":"10.1145\/1390156.1390294"},{"key":"245_CR76","doi-asserted-by":"crossref","unstructured":"Rifai S, Vincent P, Muller X, Glorot X, Bengio Y (2011) Contractive auto-encoders: explicit invariance during feature extraction. In: ICML","DOI":"10.1007\/978-3-642-23783-6_41"},{"key":"245_CR77","doi-asserted-by":"crossref","unstructured":"Zhang R, Isola P, Efros AA (2017) Split-brain autoencoders: Unsupervised learning by cross-channel prediction. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1058\u20131067","DOI":"10.1109\/CVPR.2017.76"},{"key":"245_CR78","doi-asserted-by":"crossref","unstructured":"Hinton GE, Krizhevsky A, Wang SD (2011) Transforming auto-encoders. In: International conference on artificial neural networks. Springer, pp 44\u201351","DOI":"10.1007\/978-3-642-21735-7_6"},{"key":"245_CR79","doi-asserted-by":"crossref","unstructured":"Wang F, Liu H (2021) Understanding the behaviour of contrastive loss. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2495\u20132504","DOI":"10.1109\/CVPR46437.2021.00252"},{"key":"245_CR80","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25"},{"key":"245_CR81","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556"},{"key":"245_CR82","unstructured":"Gutmann M, Hyv\u00e4rinen A (2010) Noise-contrastive estimation: A new estimation principle for unnormalized statistical models. In: Proceedings of the thirteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings, pp 297\u2013304"},{"key":"245_CR83","doi-asserted-by":"publisher","first-page":"193907","DOI":"10.1109\/ACCESS.2020.3031549","volume":"8","author":"PH Le-Khac","year":"2020","unstructured":"Le-Khac PH, Healy G, Smeaton AF (2020) Contrastive representation learning: a framework and review. IEEE Access 8:193907\u2013193934","journal-title":"IEEE Access"},{"issue":"1","key":"245_CR84","doi-asserted-by":"publisher","first-page":"2","DOI":"10.3390\/technologies9010002","volume":"9","author":"A Jaiswal","year":"2020","unstructured":"Jaiswal A, Babu AR, Zadeh MZ, Banerjee D, Makedon F (2020) A survey on contrastive self-supervised learning. Technologies 9(1):2","journal-title":"Technologies"},{"key":"245_CR85","doi-asserted-by":"crossref","unstructured":"Wu Z, Xiong Y, Yu SX, Lin D (2018) Unsupervised feature learning via non-parametric instance discrimination. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3733\u20133742","DOI":"10.1109\/CVPR.2018.00393"},{"issue":"3","key":"245_CR86","first-page":"4","volume":"2","author":"P Velickovic","year":"2019","unstructured":"Velickovic P, Fedus W, Hamilton WL, Li\u00f2 P, Bengio Y, Hjelm RD (2019) Deep graph infomax. ICLR (Poster) 2(3):4","journal-title":"ICLR (Poster)"},{"key":"245_CR87","unstructured":"Hjelm RD, Fedorov A, Lavoie-Marchildon S, Grewal K, Bachman P, Trischler A, Bengio Y (2018) Learning deep representations by mutual information estimation and maximization. arXiv:1808.06670"},{"key":"245_CR88","unstructured":"Bachman P, Hjelm RD, Buchwalter W (2019) Learning representations by maximizing mutual information across views. Adv Neural Inf Process Syst 32"},{"key":"245_CR89","unstructured":"Hassani K, Khasahmadi AH (2020) Contrastive multi-view representation learning on graphs. In: International conference on machine learning. PMLR, pp 4116\u20134126"},{"key":"245_CR90","unstructured":"Tschannen M, Djolonga J, Rubenstein PK, Gelly S, Lucic M (2019) On mutual information maximization for representation learning. arXiv:1907.13625"},{"key":"245_CR91","doi-asserted-by":"crossref","unstructured":"He K, Fan H, Wu Y, Xie S, Girshick R (2020) Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9729\u20139738","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"245_CR92","doi-asserted-by":"crossref","unstructured":"Noroozi M, Vinjimoor A, Favaro P, Pirsiavash H (2018) Boosting self-supervised learning via knowledge transfer. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 9359\u20139367","DOI":"10.1109\/CVPR.2018.00975"},{"key":"245_CR93","doi-asserted-by":"crossref","unstructured":"Tian Y, Krishnan D, Isola P (2020) Contrastive multiview coding. In: European conference on computer vision. Springer, pp 776\u2013794","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"245_CR94","first-page":"18661","volume":"33","author":"P Khosla","year":"2020","unstructured":"Khosla P, Teterwak P, Wang C, Sarna A, Tian Y, Isola P, Maschinot A, Liu C, Krishnan D (2020) Supervised contrastive learning. Adv Neural Inf Process Syst 33:18661\u201318673","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR95","doi-asserted-by":"crossref","unstructured":"Singh B, Davis LS (2018) An analysis of scale invariance in object detection snip. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3578\u20133587","DOI":"10.1109\/CVPR.2018.00377"},{"key":"245_CR96","first-page":"3407","volume":"33","author":"S Purushwalkam","year":"2020","unstructured":"Purushwalkam S, Gupta A (2020) Demystifying contrastive self-supervised learning: invariances, augmentations and dataset biases. Adv Neural Inf Process Syst 33:3407\u20133418","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR97","doi-asserted-by":"crossref","unstructured":"Giorgi J, Nitski O, Wang B, Bader G (2020) Declutr: deep contrastive learning for unsupervised textual representations. arXiv:2006.03659","DOI":"10.18653\/v1\/2021.acl-long.72"},{"key":"245_CR98","doi-asserted-by":"crossref","unstructured":"Fang H, Wang S, Zhou M, Ding J, Xie P (2020) Cert: contrastive self-supervised learning for language understanding. arXiv:2005.12766","DOI":"10.36227\/techrxiv.12308378.v1"},{"key":"245_CR99","first-page":"6256","volume":"33","author":"Q Xie","year":"2020","unstructured":"Xie Q, Dai Z, Hovy E, Luong T, Le Q (2020) Unsupervised data augmentation for consistency training. Adv Neural Inf Process Syst 33:6256\u20136268","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR100","unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013) Efficient estimation of word representations in vector space. arXiv:1301.3781"},{"key":"245_CR101","doi-asserted-by":"crossref","unstructured":"Gao T, Yao X, Chen D (2021) Simcse: simple contrastive learning of sentence embeddings. arXiv:2104.08821","DOI":"10.18653\/v1\/2021.emnlp-main.552"},{"key":"245_CR102","doi-asserted-by":"crossref","unstructured":"Yan Y, Li R, Wang S, Zhang F, Wu W, Xu W (2021) Consert: a contrastive framework for self-supervised sentence representation transfer. arXiv:2105.11741","DOI":"10.18653\/v1\/2021.acl-long.393"},{"key":"245_CR103","doi-asserted-by":"crossref","unstructured":"Rozsa A, Rudd EM, Boult TE (2016) Adversarial diversity and hard positive generation. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops, pp 25\u201332","DOI":"10.1109\/CVPRW.2016.58"},{"key":"245_CR104","doi-asserted-by":"publisher","unstructured":"Ilharco G, Zellers R, Farhadi A, Hajishirzi H (2020) Probing Contextual Language Models for Common Ground with Visual Representations. https:\/\/doi.org\/10.48550\/arxiv.2005.00619","DOI":"10.48550\/arxiv.2005.00619"},{"key":"245_CR105","unstructured":"Sun C, Baradel F, Murphy K, Schmid C (2019) Learning video representations using contrastive bidirectional transformer. arXiv:1906.05743"},{"key":"245_CR106","doi-asserted-by":"crossref","unstructured":"Senocak A, Oh T-H, Kim J, Yang M-H, Kweon IS (2018) Learning to localize sound source in visual scenes. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4358\u20134366","DOI":"10.1109\/CVPR.2018.00458"},{"issue":"5","key":"245_CR107","doi-asserted-by":"publisher","first-page":"1605","DOI":"10.1109\/TPAMI.2019.2952095","volume":"43","author":"A Senocak","year":"2019","unstructured":"Senocak A, Oh T-H, Kim J, Yang M-H, Kweon IS (2019) Learning to localize sound sources in visual scenes: analysis and applications. IEEE Trans Pattern Anal Mach Intell 43(5):1605\u20131619","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"245_CR108","doi-asserted-by":"crossref","unstructured":"Qian R, Hu D, Dinkel H, Wu M, Xu N, Lin W (2020) Multiple sound sources localization from coarse to fine. In: European conference on computer vision. Springer, pp 292\u2013308","DOI":"10.1007\/978-3-030-58565-5_18"},{"key":"245_CR109","doi-asserted-by":"crossref","unstructured":"Hu D, Nie F, Li X (2019) Deep multimodal clustering for unsupervised audiovisual learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9248\u20139257","DOI":"10.1109\/CVPR.2019.00947"},{"key":"245_CR110","first-page":"10077","volume":"33","author":"D Hu","year":"2020","unstructured":"Hu D, Qian R, Jiang M, Tan X, Wen S, Ding E, Lin W, Dou D (2020) Discriminative sounding objects localization via self-supervised audiovisual matching. Adv Neural Inf Process Syst 33:10077\u201310087","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR111","unstructured":"Hu D, Wang Z, Xiong H, Wang D, Nie F, Dou D (2020) Curriculum audiovisual learning. arXiv:2001.09414"},{"key":"245_CR112","doi-asserted-by":"crossref","unstructured":"Zhan X, Xie J, Liu Z, Ong Y-S, Loy CC (2020) Online deep clustering for unsupervised representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6688\u20136697","DOI":"10.1109\/CVPR42600.2020.00672"},{"key":"245_CR113","unstructured":"Tao Y, Takagi K, Nakata K (2021) Clustering-friendly representation learning via instance discrimination and feature decorrelation. arXiv:2106.00131"},{"key":"245_CR114","unstructured":"Tsai TW, Li C, Zhu J (2020) Mice: mixture of contrastive experts for unsupervised image clustering. In: International conference on learning representations"},{"key":"245_CR115","doi-asserted-by":"crossref","unstructured":"Hu Q, Wang X, Hu W, Qi G-J (2021) Adco: adversarial contrast for efficient learning of unsupervised representations from self-trained negative adversaries. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1074\u20131083","DOI":"10.1109\/CVPR46437.2021.00113"},{"key":"245_CR116","unstructured":"Chen X, Fan H, Girshick R, He K (2020) Improved baselines with momentum contrastive learning. arXiv:2003.04297"},{"key":"245_CR117","first-page":"21798","volume":"33","author":"Y Kalantidis","year":"2020","unstructured":"Kalantidis Y, Sariyildiz MB, Pion N, Weinzaepfel P, Larlus D (2020) Hard negative mixing for contrastive learning. Adv Neural Inf Process Syst 33:21798\u201321809","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR118","unstructured":"Robinson J, Chuang C-Y, Sra S, Jegelka S (2020) Contrastive learning with hard negative samples. arXiv:2010.04592"},{"key":"245_CR119","unstructured":"Sohn K (2016) Improved deep metric learning with multi-class n-pair loss objective. Adv Neural Inf Process Syst 29"},{"key":"245_CR120","doi-asserted-by":"crossref","unstructured":"Wu C, Wu F, Huang Y (2021) Rethinking infonce: How many negative samples do you need? arXiv:2105.13003","DOI":"10.24963\/ijcai.2022\/348"},{"key":"245_CR121","doi-asserted-by":"crossref","unstructured":"Schroff F, Kalenichenko D, Philbin J (2015) Facenet: a unified embedding for face recognition and clustering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 815\u2013823","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"245_CR122","doi-asserted-by":"crossref","unstructured":"Wang X, Hua Y, Kodirov E, Hu G, Garnier R, Robertson NM (2019) Ranked list loss for deep metric learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5207\u20135216","DOI":"10.1109\/CVPR.2019.00535"},{"key":"245_CR123","unstructured":"Weinberger KQ, Saul LK (2009) Distance metric learning for large margin nearest neighbor classification. J Mach Learn Res 10(2)"},{"key":"245_CR124","doi-asserted-by":"crossref","unstructured":"Chopra S, Hadsell R, LeCun Y (2005) Learning a similarity metric discriminatively, with application to face verification. In: 2005 IEEE computer society conference on computer vision and pattern recognition (CVPR\u201905), vol\u00a01. IEEE, pp 539\u2013546","DOI":"10.1109\/CVPR.2005.202"},{"key":"245_CR125","doi-asserted-by":"crossref","unstructured":"Hadsell R, Chopra S, LeCun Y (2006) Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE computer society conference on computer vision and pattern recognition (CVPR\u201906), vol\u00a02. IEEE, pp 1735\u20131742","DOI":"10.1109\/CVPR.2006.100"},{"key":"245_CR126","doi-asserted-by":"crossref","unstructured":"Oh\u00a0Song H, Xiang Y, Jegelka S, Savarese S (2016) Deep metric learning via lifted structured feature embedding. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4004\u20134012","DOI":"10.1109\/CVPR.2016.434"},{"key":"245_CR127","unstructured":"Goldberger J, Hinton G\u00a0E, Roweis S, Salakhutdinov R\u00a0R, \u201cNeighbourhood components analysis,\u201d Advances in neural information processing systems, vol.\u00a017, (2004)"},{"key":"245_CR128","unstructured":"Ghojogh B, Karray F, Crowley M (2019) Fisher and kernel fisher discriminant analysis: tutorial. arXiv:1906.09436"},{"key":"245_CR129","unstructured":"Sun Z, Deng Z-H, Nie J-Y, Tang J (2019) Rotate: knowledge graph embedding by relational rotation in complex space. arXiv:1902.10197"},{"key":"245_CR130","first-page":"1727","volume":"2021","author":"Z Li","year":"2021","unstructured":"Li Z, Ji J, Fu Z, Ge Y, Xu S, Chen C, Zhang Y (2021) Efficient non-sampling knowledge graph embedding. Proc Web Conf 2021:1727\u20131736","journal-title":"Proc Web Conf"},{"key":"245_CR131","doi-asserted-by":"crossref","unstructured":"Peng X, Chen G, Lin C, Stevenson M (2021) Highly efficient knowledge graph embedding learning with orthogonal procrustes analysis. arXiv:2104.04676","DOI":"10.18653\/v1\/2021.naacl-main.187"},{"key":"245_CR132","unstructured":"Cheng JY, Goh H, Dogrusoz K, Tuzel O, Azemi E (2020) Subject-aware contrastive learning for biosignals. arXiv:2007.04871"},{"issue":"6356","key":"245_CR133","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1038\/355161a0","volume":"355","author":"S Becker","year":"1992","unstructured":"Becker S, Hinton GE (1992) Self-organizing neural network that discovers surfaces in random-dot stereograms. Nature 355(6356):161\u2013163","journal-title":"Nature"},{"key":"245_CR134","doi-asserted-by":"crossref","unstructured":"Bromley J, Guyon I, LeCun Y, S\u00e4ckinger E, Shah R (1993) Signature verification using a \u201csiamese\u201d time delay neural network. Adv Neural Inf Process Syst 6","DOI":"10.1142\/9789812797926_0003"},{"key":"245_CR135","doi-asserted-by":"crossref","unstructured":"Chi Z, Dong L, Wei F, Yang N, Singhal S, Wang W, Song X, Mao X-L, Huang H, Zhou M (2020) Infoxlm: an information-theoretic framework for cross-lingual language model pre-training. arXiv:2007.07834","DOI":"10.18653\/v1\/2021.naacl-main.280"},{"key":"245_CR136","unstructured":"Lample G, Conneau A (2019) Cross-lingual language model pretraining. arXiv:1901.07291"},{"key":"245_CR137","unstructured":"Wu Z, Wang S, Gu J, Khabsa M, Sun F, Ma H (2020) Clear: contrastive learning for sentence representation. arXiv:2012.15466"},{"key":"245_CR138","doi-asserted-by":"crossref","unstructured":"Wei J, Zou K (2019) Eda: easy data augmentation techniques for boosting performance on text classification tasks. arXiv:1901.11196","DOI":"10.18653\/v1\/D19-1670"},{"key":"245_CR139","unstructured":"Liao D (2021) Sentence embeddings using supervised contrastive learning. arXiv:2106.04791"},{"key":"245_CR140","unstructured":"Arora S, Khandeparkar H, Khodak M, Plevrakis O, Saunshi N (2019) A theoretical analysis of contrastive unsupervised representation learning. arXiv:1902.09229"},{"key":"245_CR141","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013) Distributed representations of words and phrases and their compositionality. Adv Neural Inf Process Syst 26"},{"key":"245_CR142","doi-asserted-by":"crossref","unstructured":"Simoulin A, Crabb\u00e9 B (2021) Contrasting distinct structured views to learn sentence embeddings. In: European chapter of the association of computational linguistics (student)","DOI":"10.18653\/v1\/2021.eacl-srw.11"},{"key":"245_CR143","doi-asserted-by":"crossref","unstructured":"Aroca-Ouellette S, Rudzicz F (2020) On losses for modern language models. arXiv:2010.01694","DOI":"10.18653\/v1\/2020.emnlp-main.403"},{"key":"245_CR144","doi-asserted-by":"crossref","unstructured":"Sun S, Gan Z, Cheng Y, Fang Y, Wang S, Liu J (2020) Contrastive distillation on intermediate representations for language model compression. arXiv:2009.14167","DOI":"10.18653\/v1\/2020.emnlp-main.36"},{"key":"245_CR145","unstructured":"Deng Y, Bakhtin A, Ott M, Szlam A, Ranzato M (2020) Residual energy-based models for text generation. arXiv:2004.11714"},{"key":"245_CR146","unstructured":"Lai C-I (2019) Contrastive predictive coding based feature for automatic speaker verification. arXiv:1904.01575"},{"key":"245_CR147","unstructured":"Zhang S, Yan J, Yang X (2020) Self-supervised representation learning via adaptive hard-positive mining"},{"key":"245_CR148","doi-asserted-by":"crossref","unstructured":"Huynh T, Kornblith S, Walter MR, Maire M, Khademi M (2022) Boosting contrastive self-supervised learning with false negative cancellation. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 2785\u20132795","DOI":"10.1109\/WACV51458.2022.00106"},{"key":"245_CR149","unstructured":"Ermolov A, Siarohin A, Sangineto E, Sebe N (2021) Whitening for self-supervised representation learning. In: International conference on machine learning. PMLR, pp 3015\u20133024"},{"key":"245_CR150","doi-asserted-by":"crossref","unstructured":"Yao Y, Liu C, Luo D, Zhou Y, Ye Q (2020) Video playback rate perception for self-supervised spatio-temporal representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6548\u20136557","DOI":"10.1109\/CVPR42600.2020.00658"},{"key":"245_CR151","unstructured":"Bai Y, Fan H, Misra I, Venkatesh G, Lu Y, Zhou Y, Yu Q, Chandra V, Yuille A (2020) Can temporal information help with contrastive self-supervised learning? arXiv:2011.13046"},{"key":"245_CR152","doi-asserted-by":"crossref","unstructured":"Pan T, Song Y, Yang T, Jiang W, Liu W (2021) Videomoco: contrastive video representation learning with temporally adversarial examples. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11205\u201311214","DOI":"10.1109\/CVPR46437.2021.01105"},{"key":"245_CR153","unstructured":"Yang C, Xu Y, Dai B, Zhou B (2020) Video representation learning with visual tempo consistency. arXiv:2006.15489"},{"key":"245_CR154","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Fan H, Malik J, He K (2019) Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 6202\u20136211","DOI":"10.1109\/ICCV.2019.00630"},{"key":"245_CR155","doi-asserted-by":"crossref","unstructured":"Zhuang C, She T, Andonian A, Mark M\u00a0S, Yamins D (2020) Unsupervised learning from video with deep neural embeddings. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9563\u20139572","DOI":"10.1109\/CVPR42600.2020.00958"},{"key":"245_CR156","doi-asserted-by":"crossref","unstructured":"Han T, Xie W, Zisserman A (2019) Video representation learning by dense predictive coding. In: Proceedings of the IEEE\/CVF international conference on computer vision workshops","DOI":"10.1109\/ICCVW.2019.00186"},{"key":"245_CR157","doi-asserted-by":"crossref","unstructured":"Han T, Xie W, Zisserman A (2020) Memory-augmented dense predictive coding for video representation learning. In: European conference on computer vision. Springer, pp 312\u2013329","DOI":"10.1007\/978-3-030-58580-8_19"},{"key":"245_CR158","doi-asserted-by":"crossref","unstructured":"Lorre G, Rabarisoa J, Orcesi A, Ainouz S, Canu S (2020) Temporal contrastive pretraining for video action recognition. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 662\u2013670","DOI":"10.1109\/WACV45572.2020.9093278"},{"key":"245_CR159","doi-asserted-by":"crossref","unstructured":"Caron M, Bojanowski P, Joulin A, Douze M (2018) Deep clustering for unsupervised learning of visual features. In: Proceedings of the European conference on computer vision (ECCV), pp 132\u2013149","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"245_CR160","doi-asserted-by":"crossref","unstructured":"Zhuang C, Zhai A\u00a0L, Yamins D (2019) Local aggregation for unsupervised learning of visual embeddings. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 6002\u20136012","DOI":"10.1109\/ICCV.2019.00610"},{"key":"245_CR161","unstructured":"Li J, Zhou P, Xiong C, Hoi SC (2020) Prototypical contrastive learning of unsupervised representations. arXiv:2005.04966"},{"key":"245_CR162","unstructured":"Hjelm RD, Bachman P (2020) Representation learning with video deep infomax. arXiv:2007.13278"},{"key":"245_CR163","doi-asserted-by":"publisher","DOI":"10.1016\/j.image.2020.115967","volume":"88","author":"F Xue","year":"2020","unstructured":"Xue F, Ji H, Zhang W, Cao Y (2020) Self-supervised video representation learning by maximizing mutual information. Signal Process Image Commun 88:115967","journal-title":"Signal Process Image Commun"},{"key":"245_CR164","doi-asserted-by":"crossref","unstructured":"Wang J, Jiao J, Liu Y-H (2020) Self-supervised video representation learning by pace prediction. In: European conference on computer vision. Springer, pp 504\u2013521","DOI":"10.1007\/978-3-030-58520-4_30"},{"key":"245_CR165","doi-asserted-by":"crossref","unstructured":"Knights J, Harwood B, Ward D, Vanderkop A, Mackenzie-Ross O, Moghadam P (2021) Temporally coherent embeddings for self-supervised video representation learning. In: 2020 25th international conference on pattern recognition (ICPR). IEEE, pp 8914\u20138921","DOI":"10.1109\/ICPR48806.2021.9412071"},{"key":"245_CR166","doi-asserted-by":"crossref","unstructured":"Yao T, Zhang Y, Qiu Z, Pan Y, Mei T (2021) Seco: exploring sequence supervision for unsupervised representation learning. In: AAAI, vol\u00a02, p\u00a07","DOI":"10.1609\/aaai.v35i12.17274"},{"key":"245_CR167","doi-asserted-by":"crossref","unstructured":"Tao L, Wang X, Yamasaki T (2020) Self-supervised video representation learning using inter-intra contrastive framework. In: Proceedings of the 28th ACM international conference on multimedia, pp 2193\u20132201","DOI":"10.1145\/3394171.3413694"},{"key":"245_CR168","unstructured":"Wang J, Gao Y, Li K, Jiang X, Guo X, Ji R, Sun X (2021) Enhancing unsupervised video representation learning by decoupling the scene and the motion. In: AAAI, vol\u00a01, no.\u00a02, p\u00a07"},{"key":"245_CR169","doi-asserted-by":"crossref","unstructured":"Afouras T, Owens A, Chung JS, Zisserman A (2020) Self-supervised learning of audio-visual objects from video. In: European conference on computer vision. Springer, pp 208\u2013224","DOI":"10.1007\/978-3-030-58523-5_13"},{"key":"245_CR170","doi-asserted-by":"crossref","unstructured":"Miech A, Alayrac J-B, Smaira L, Laptev I, Sivic J, Zisserman A (2020) End-to-end learning of visual representations from uncurated instructional videos. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9879\u20139889","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"245_CR171","doi-asserted-by":"crossref","unstructured":"Tokmakov P, Hebert M, Schmid C (2020) Unsupervised learning of video representations via dense trajectory clustering. In: European conference on computer vision. Springer, pp 404\u2013421","DOI":"10.1007\/978-3-030-66096-3_28"},{"key":"245_CR172","doi-asserted-by":"crossref","unstructured":"Dunbar E, Karadayi J, Bernard M, Cao X-N, Algayres R, Ondel L, Besacier L, Sakti S, Dupoux E (2020) The zero resource speech challenge 2020: discovering discrete subword and word units. arXiv:2010.05967","DOI":"10.21437\/Interspeech.2020-2743"},{"key":"245_CR173","doi-asserted-by":"crossref","unstructured":"Glass J (2012) Towards unsupervised speech processing. In: 2012 11th international conference on information science, signal processing and their applications (ISSPA). IEEE, pp 1\u20134","DOI":"10.1109\/ISSPA.2012.6310546"},{"key":"245_CR174","unstructured":"Schatz T (2016) Abx-discriminability measures and applications. Ph.D. Dissertation, Universit\u00e9 Paris 6 (UPMC)"},{"key":"245_CR175","doi-asserted-by":"crossref","unstructured":"Dunbar E, Cao XN, Benjumea J, Karadayi J, Bernard M, Besacier L, Anguera X, Dupoux E (2017) The zero resource speech challenge 2017. In: 2017 IEEE automatic speech recognition and understanding workshop (ASRU). IEEE, pp 323\u2013330","DOI":"10.1109\/ASRU.2017.8268953"},{"key":"245_CR176","unstructured":"Kawakami K, Wang L, Dyer C, Blunsom P, van der Oord A: Learning robust and multilingual speech representations. arXiv:2001.11128"},{"key":"245_CR177","doi-asserted-by":"crossref","unstructured":"Wang W, Tang Q, Livescu K (2020) Unsupervised pre-training of bidirectional speech encoders via masked reconstruction. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6889\u20136893","DOI":"10.1109\/ICASSP40776.2020.9053541"},{"key":"245_CR178","doi-asserted-by":"crossref","unstructured":"Heck M, Sakti S, Nakamura S (2017) Feature optimized DPGMM clustering for unsupervised subword modeling: A contribution to zerospeech 2017. In: 2017 IEEE automatic speech recognition and understanding workshop (ASRU). IEEE, pp 740\u2013746","DOI":"10.1109\/ASRU.2017.8269011"},{"key":"245_CR179","unstructured":"Nandan A, Vepa J (2020) Language agnostic speech embeddings for emotion classification"},{"key":"245_CR180","doi-asserted-by":"crossref","unstructured":"Park DS, Chan W, Zhang Y, Chiu C-C, Zoph B, Cubuk ED, Le QV (2019) Specaugment: a simple data augmentation method for automatic speech recognition. arXiv:1904.08779","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"245_CR181","doi-asserted-by":"crossref","unstructured":"Shor J, Jansen A, Han W, Park D, Zhang Y (2021) Universal paralinguistic speech representations using self-supervised conformers. arXiv:2110.04621","DOI":"10.1109\/ICASSP43922.2022.9747197"},{"key":"245_CR182","unstructured":"Al-Tahan H, Mohsenzadeh Y (2021) Clar: contrastive learning of auditory representations. In: International conference on artificial intelligence and statistics. PMLR, pp 2530\u20132538"},{"key":"245_CR183","doi-asserted-by":"crossref","unstructured":"Saeed A, Grangier D, Zeghidour N (2021) Contrastive learning of general-purpose audio representations. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 3875\u20133879","DOI":"10.1109\/ICASSP39728.2021.9413528"},{"key":"245_CR184","doi-asserted-by":"crossref","unstructured":"Xia J, Wu L, Chen J, Hu B, Li SZ (2022) Simgrace: a simple framework for graph contrastive learning without data augmentation. arXiv:2202.03104","DOI":"10.1145\/3485447.3512156"},{"key":"245_CR185","unstructured":"Wang T, Isola P (2020) Understanding contrastive representation learning through alignment and uniformity on the hypersphere. In: International conference on machine learning. PMLR, pp 9929\u20139939"},{"key":"245_CR186","unstructured":"You Y, Chen T, Shen Y, Wang Z (2021) Graph contrastive learning automated. In: International conference on machine learning. PMLR, pp 12121\u201312132"},{"key":"245_CR187","unstructured":"Zeng J, Xie P (2020) Contrastive self-supervised learning for graph classification. arXiv:2009.05923"},{"key":"245_CR188","first-page":"5812","volume":"33","author":"Y You","year":"2020","unstructured":"You Y, Chen T, Sui Y, Chen T, Wang Z, Shen Y (2020) Graph contrastive learning with augmentations. Adv Neural Inf Process Syst 33:5812\u20135823","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR189","unstructured":"Sun M, Xing J, Wang H, Chen B, Zhou J, \u201cMocl: Contrastive learning on molecular graphs with multi-level domain knowledge,\u201d arXiv preprint arXiv:2106.04509, (2021)"},{"key":"245_CR190","unstructured":"Sun F-Y, Hoffmann J, Verma V, Tang J (2019) Infograph: Unsupervised and semi-supervised graph-level representation learning via mutual information maximization. arXiv:1908.01000"},{"key":"245_CR191","first-page":"2069","volume":"2021","author":"Y Zhu","year":"2021","unstructured":"Zhu Y, Xu Y, Yu F, Liu Q, Wu S, Wang L (2021) Graph contrastive learning with adaptive augmentation. Proc Web Conf 2021:2069\u20132080","journal-title":"Proc Web Conf"},{"key":"245_CR192","unstructured":"Xia J, Wu L, Chen J, Wang G, Li SZ (2021) Debiased graph contrastive learning. arXiv:2110.02027"},{"key":"245_CR193","first-page":"25","volume":"33","author":"J-B Alayrac","year":"2020","unstructured":"Alayrac J-B, Recasens A, Schneider R, Arandjelovi\u0107 R, Ramapuram J, De Fauw J, Smaira L, Dieleman S, Zisserman A (2020) Self-supervised multimodal versatile networks. Adv Neural Inf Process Syst 33:25\u201337","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR194","unstructured":"Liu Y, Yi L, Zhang S, Fan Q, Funkhouser T, Dong H (2020) P4contrast: contrastive learning with pairs of point-pixel pairs for RGB-D scene understanding. arXiv:2012.13089"},{"key":"245_CR195","first-page":"8765","volume":"33","author":"C-Y Chuang","year":"2020","unstructured":"Chuang C-Y, Robinson J, Lin Y-C, Torralba A, Jegelka S (2020) Debiased contrastive learning. Adv Neural Inf Process Syst 33:8765\u20138775","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR196","first-page":"17081","volume":"33","author":"C-H Ho","year":"2020","unstructured":"Ho C-H, Nvasconcelos N (2020) Contrastive learning with adversarial examples. Adv Neural Inf Process Syst 33:17081\u201317093","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR197","first-page":"6827","volume":"33","author":"Y Tian","year":"2020","unstructured":"Tian Y, Sun C, Poole B, Krishnan D, Schmid C, Isola P (2020) What makes for good views for contrastive learning? Adv Neural Inf Process Syst 33:6827\u20136839","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR198","unstructured":"Wu M, Zhuang C, Mosse M, Yamins D, Goodman N (2020) On mutual information in contrastive learning for visual representations. arXiv:2005.13149"},{"key":"245_CR199","first-page":"4660","volume":"33","author":"Y Asano","year":"2020","unstructured":"Asano Y, Patrick M, Rupprecht C, Vedaldi A (2020) Labelling unlabelled videos from scratch with multi-modal self-supervision. Adv Neural Inf Process Syst 33:4660\u20134671","journal-title":"Adv Neural Inf Process Syst"},{"key":"245_CR200","doi-asserted-by":"crossref","unstructured":"Morgado P, Vasconcelos N, Misra I (2021) Audio-visual instance discrimination with cross-modal agreement. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12475\u201312486","DOI":"10.1109\/CVPR46437.2021.01229"},{"key":"245_CR201","unstructured":"Patrick M, Asano YM, Kuznetsova P, Fong R, Henriques JF, Zweig G, Vedaldi A (2020) Multi-modal self-supervision from generalized data transformations. arXiv:2003.04298"},{"key":"245_CR202","unstructured":"Xiao F, Lee YJ, Grauman K, Malik J, Feichtenhofer C (2020) Audiovisual slowfast networks for video recognition. arXiv:2001.08740"},{"key":"245_CR203","doi-asserted-by":"crossref","unstructured":"Gan C, Huang D, Zhao H, Tenenbaum JB, Torralba A (2020) Music gesture for visual sound separation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10478\u201310487","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"245_CR204","doi-asserted-by":"crossref","unstructured":"Yang K, Russell B, Salamon J (2020) Telling left from right: learning spatial correspondence of sight and sound. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9932\u20139941","DOI":"10.1109\/CVPR42600.2020.00995"},{"key":"245_CR205","unstructured":"Lin Y-B, Tseng H-Y, Lee H-Y, Lin Y-Y, Yang M-H (2021) Unsupervised sound localization via iterative contrastive learning. arXiv:2104.00315"},{"key":"245_CR206","doi-asserted-by":"crossref","unstructured":"Nagrani A, Chung JS, Albanie S, Zisserman A (2020) Disentangled speech embeddings using cross-modal self-supervision. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6829\u20136833","DOI":"10.1109\/ICASSP40776.2020.9054057"},{"key":"245_CR207","doi-asserted-by":"crossref","unstructured":"Li B, Zhou H, He J, Wang M, Yang Y, Li L (2020) On the sentence embeddings from pre-trained language models. arXiv:2011.05864","DOI":"10.18653\/v1\/2020.emnlp-main.733"},{"key":"245_CR208","doi-asserted-by":"crossref","unstructured":"Reimers N, Gurevych I (2019) Sentence-Bert: sentence embeddings using Siamese Bert-networks. arXiv:1908.10084","DOI":"10.18653\/v1\/D19-1410"},{"key":"245_CR209","doi-asserted-by":"crossref","unstructured":"Jain P, Jain A, Zhang T, Abbeel P, Gonzalez JE, Stoica I (2020) Contrastive code representation learning. arXiv:2007.04973","DOI":"10.18653\/v1\/2021.emnlp-main.482"},{"key":"245_CR210","doi-asserted-by":"crossref","unstructured":"Bui N\u00a0D, Yu Y, Jiang L (2021) Self-supervised contrastive learning for code retrieval and summarization via semantic-preserving transformations. In: Proceedings of the 44th International ACM SIGIR conference on research and development in information retrieval, pp 511\u2013521","DOI":"10.1145\/3404835.3462840"},{"key":"245_CR211","doi-asserted-by":"crossref","unstructured":"Li Y, Hu P, Liu Z, Peng D, Zhou JT, Peng X (2021) Contrastive clustering. In: 2021 AAAI conference on artificial intelligence (AAAI)","DOI":"10.1609\/aaai.v35i10.17037"},{"key":"245_CR212","doi-asserted-by":"crossref","unstructured":"Lin Y, Gou Y, Liu Z, Li B, Lv J, Peng X (2021) Completer: incomplete multi-view clustering via contrastive prediction. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11174\u201311183","DOI":"10.1109\/CVPR46437.2021.01102"},{"key":"245_CR213","unstructured":"Pan E, Kang Z (2021) Multi-view contrastive graph clustering. Adv Neural Inf Process Syst 34"},{"key":"245_CR214","doi-asserted-by":"crossref","unstructured":"Trosten DJ, Lokse S, Jenssen R, Kampffmeyer M (2021) Reconsidering representation alignment for multi-view clustering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1255\u20131265","DOI":"10.1109\/CVPR46437.2021.00131"},{"key":"245_CR215","doi-asserted-by":"crossref","unstructured":"Wu L, Lin H, Tan C, Gao Z, Li SZ (2021) Self-supervised learning on graphs: contrastive, generative, or predictive. IEEE Trans Knowl Data Eng","DOI":"10.1109\/TKDE.2021.3131584"},{"key":"245_CR216","doi-asserted-by":"crossref","unstructured":"Bhattacharjee A, Karami M, Liu H (2022) Text transformations in contrastive self-supervised learning: a review. arXiv:2203.12000","DOI":"10.24963\/ijcai.2022\/757"},{"issue":"4","key":"245_CR217","doi-asserted-by":"publisher","first-page":"551","DOI":"10.3390\/e24040551","volume":"24","author":"S Albelwi","year":"2022","unstructured":"Albelwi S (2022) Survey on self-supervised learning: auxiliary pretext tasks and contrastive learning methods in imaging. Entropy 24(4):551","journal-title":"Entropy"},{"key":"245_CR218","unstructured":"Stephane A-O, Frank R (2020) On losses for modern language models. arXiv:2010.01694"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00245-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00245-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00245-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,17]],"date-time":"2022-12-17T14:22:12Z","timestamp":1671286932000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00245-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,5]]},"references-count":218,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["245"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00245-6","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,5]]},"assertion":[{"value":"30 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 July 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 August 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Yes.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}}]}}