{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T17:21:31Z","timestamp":1782408091802,"version":"3.54.5"},"reference-count":76,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Nature Science Foundation of China","doi-asserted-by":"publisher","award":["62372155"],"award-info":[{"award-number":["62372155"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Joint Fund of the Ministry of Education for Equipment Preresearch","award":["8091B022123"],"award-info":[{"award-number":["8091B022123"]}]},{"name":"Research Fund from Science and Technology on Underwater Vehicle Technology Laboratory","award":["2021JCJQ-SYSJJ-LB06905"],"award-info":[{"award-number":["2021JCJQ-SYSJJ-LB06905"]}]},{"name":"Key Laboratory of Information System Requirements","award":["LHZZ 2021-M04"],"award-info":[{"award-number":["LHZZ 2021-M04"]}]},{"DOI":"10.13039\/501100013088","name":"Water Science and Technology Project of Jiangsu Province","doi-asserted-by":"publisher","award":["2021063"],"award-info":[{"award-number":["2021063"]}],"id":[{"id":"10.13039\/501100013088","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1109\/tnnls.2023.3335859","type":"journal-article","created":{"date-parts":[[2023,12,4]],"date-time":"2023-12-04T13:55:52Z","timestamp":1701698152000},"page":"610-624","source":"Crossref","is-referenced-by-count":43,"title":["ProtoCLIP: Prototypical Contrastive Language Image Pretraining"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8172-2894","authenticated-orcid":false,"given":"Delong","family":"Chen","sequence":"first","affiliation":[{"name":"College of Computer and Information, Hohai University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhao","family":"Wu","sequence":"additional","affiliation":[{"name":"MEGVII Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8746-9845","authenticated-orcid":false,"given":"Fan","family":"Liu","sequence":"additional","affiliation":[{"name":"College of Computer and Information, Hohai University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zaiquan","family":"Yang","sequence":"additional","affiliation":[{"name":"MEGVII Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaoqiu","family":"Zheng","sequence":"additional","affiliation":[{"name":"Nanjing Research Institute of Electronic Engineering, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8243-4731","authenticated-orcid":false,"given":"Ying","family":"Tan","sequence":"additional","affiliation":[{"name":"Department of Machine Intelligence, School of EECS, Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Erjin","family":"Zhou","sequence":"additional","affiliation":[{"name":"MEGVII Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. 38th Int. Conf. Mach. Learn. (ICML)","volume":"139","author":"Radford"},{"key":"ref2","article-title":"Representation learning with contrastive predictive coding","author":"van den Oord","year":"2018","journal-title":"arXiv:1807.03748"},{"key":"ref3","first-page":"9929","article-title":"Understanding contrastive representation learning through alignment and uniformity on the hypersphere","volume-title":"Proc. 37th Int. Conf. Mach. Learn. (ICML)","volume":"119","author":"Wang"},{"key":"ref4","article-title":"Chaos is a ladder: A new theoretical understanding of contrastive learning via augmentation overlap","author":"Wang","year":"2022","journal-title":"arXiv:2203.13457"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref6","article-title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","author":"Liang","year":"2022","journal-title":"arXiv:2203.02053"},{"key":"ref7","article-title":"Lat: Latent translation with cycle-consistency for video-text retrieval","author":"Bai","year":"2022","journal-title":"arXiv:2207.04858"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1503.02531"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00302"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1907.11692"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"ref12","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NIPS)","author":"Caron"},{"key":"ref13","article-title":"Self-labelling via simultaneous clustering and representation learning","volume-title":"Proc. 8th Int. Conf. Learn. Represent. (ICLR)","author":"Asano"},{"key":"ref14","article-title":"Prototypical contrastive learning of unsupervised representations","volume-title":"Proc. 9th Int. Conf. Learn. Represent. (ICLR)","author":"Li"},{"key":"ref15","first-page":"9758","article-title":"Self-supervised learning by cross-modal audio-video clustering","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Alwassel"},{"key":"ref16","first-page":"4660","article-title":"Labelling unlabelled videos from scratch with multi-modal self-supervision","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Asano"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00791"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/2812802"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.427"},{"key":"ref21","article-title":"VisualBERT: A simple and performant baseline for vision and language","author":"Li","year":"2019","journal-title":"arXiv:1908.03557"},{"key":"ref22","first-page":"13","article-title":"ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Lu"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29727"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3227717"},{"key":"ref28","article-title":"FILIP: Fine-grained interactive language-image pre-training","author":"Yao","year":"2021","journal-title":"arXiv:2111.07783"},{"key":"ref29","article-title":"CLOOB: Modern Hopfield networks with infoloob outperform CLIP","author":"F\u00fcrst","year":"2021","journal-title":"arXiv:2110.11316"},{"key":"ref30","article-title":"EfficientCLIP: Efficient cross-modal pre-training by ensemble confident learning and language modeling","author":"Wang","year":"2021","journal-title":"arXiv:2109.04699"},{"key":"ref31","article-title":"SLIP: Self-supervision meets language-image pre-training","author":"Mu","year":"2021","journal-title":"arXiv:2112.12750"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref33","article-title":"Supervision exists everywhere: A data efficient contrastive language-image pre-training paradigm","author":"Li","year":"2021","journal-title":"arXiv:2110.05208"},{"key":"ref34","article-title":"RemoteCLIP: A vision language foundation model for remote sensing","author":"Liu","year":"2023","journal-title":"arXiv:2306.11029"},{"key":"ref35","article-title":"Self-supervised representation learning: Introduction, advances and challenges","author":"Ericsson","year":"2021","journal-title":"arXiv:2110.09327"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992393"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497510"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00945"},{"key":"ref41","article-title":"Mine your own view: Self-supervised learning through across-sample prediction","author":"Azabou","year":"2021","journal-title":"arXiv:2102.10106"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3191086"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3083650"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2023.3297607"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00409"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3188569"},{"key":"ref47","article-title":"Democratizing contrastive language-image pre-training: A CLIP benchmark of data, model, and supervision","author":"Cui","year":"2022","journal-title":"arXiv:2203.05796"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2021.3084827"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00065"},{"key":"ref51","first-page":"7324","article-title":"Making convolutional networks shift-invariant again","volume-title":"Proc. 36th Int. Conf. Mach. Learn. (ICML)","volume":"97","author":"Zhang"},{"key":"ref52","volume-title":"OpenCLIP","author":"Ilharco","year":"2021"},{"key":"ref53","article-title":"Mixed precision training","volume-title":"Proc. 6th Int. Conf. Learn. Represent. (ICLR)","author":"Micikevicius"},{"key":"ref54","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. 3rd Int. Conf. Learn. Represent. (ICLR)","author":"Kingma"},{"key":"ref55","article-title":"Decoupled weight decay regularization","volume-title":"Proc. 7th Int. Conf. Learn. Represent. (ICLR)","author":"Loshchilov"},{"key":"ref56","article-title":"SGDR: Stochastic gradient descent with warm restarts","volume-title":"Proc. 5th Int. Conf. Learn. Represent. (ICLR)","author":"Loshchilov"},{"key":"ref57","article-title":"LiT: Zero-shot transfer with locked-image text tuning","author":"Zhai","year":"2021","journal-title":"arXiv:2111.07991"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref60","first-page":"28864","article-title":"Unsupervised object-level representation learning from scene images","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst. (NIPS)","author":"Xie"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01422"},{"key":"ref62","first-page":"7187","article-title":"Investigating the role of negatives in contrastive representation learning","volume-title":"Proc. Int. Conf. Artif. Intell. Statist. (AISTATS)","volume":"151","author":"Ash"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.461"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.259"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2011.6033395"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1212.0402"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.77"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref72","article-title":"Microsoft COCO captions: Data collection and evaluation server","author":"Chen","year":"2015","journal-title":"arXiv:1504.00325"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2844175"},{"key":"ref74","article-title":"DenseCLIP: Language-guided dense prediction with context-aware prompting","author":"Rao","year":"2021","journal-title":"arXiv:2112.01518"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01857"},{"issue":"11","key":"ref76","first-page":"1","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10832116\/10339644.pdf?arnumber=10339644","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,5]],"date-time":"2025-12-05T18:39:09Z","timestamp":1764959949000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10339644\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1]]},"references-count":76,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3335859","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1]]}}}