{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T15:41:02Z","timestamp":1785858062142,"version":"3.56.0"},"reference-count":202,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1109\/tpami.2024.3369699","type":"journal-article","created":{"date-parts":[[2024,2,26]],"date-time":"2024-02-26T20:01:52Z","timestamp":1708977712000},"page":"5625-5644","source":"Crossref","is-referenced-by-count":843,"title":["Vision-Language Models for Vision Tasks: A Survey"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4519-8244","authenticated-orcid":false,"given":"Jingyi","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8681-0471","authenticated-orcid":false,"given":"Jiaxing","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7254-1664","authenticated-orcid":false,"given":"Sheng","family":"Jin","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6766-2506","authenticated-orcid":false,"given":"Shijian","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2020.3005403"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1080\/01691864.2017.1365009"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref5","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2014"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.4249\/scholarpedia.1883"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/BF00994018"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2008.915597"},{"key":"ref10","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref14","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref15","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018","journal-title":"OpenAI Blog"},{"issue":"8","key":"ref16","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref17","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Jia"},{"key":"ref18","first-page":"1","article-title":"Filip: Fine-grained interactive language-image pre-training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yao"},{"key":"ref19","article-title":"COCA: Contrastive captioners are image-text foundation models","author":"Yu","year":"2022"},{"key":"ref20","article-title":"LAION-5B: An open large-scale dataset for training next generation image-text models","author":"Schuhmann","year":"2022"},{"key":"ref21","article-title":"LAION-400M: Open dataset of clip-filtered 400 million image-text pairs","author":"Schuhmann","year":"2021"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"ref23","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"ref25","article-title":"Collecting a large-scale dataset of fine-grained cars","author":"Krause","year":"2013"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"ref27","first-page":"2611","article-title":"The hateful memes challenge: Detecting hate speech in multimodal memes","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kiela"},{"key":"ref28","article-title":"RareAct: A video dataset of unusual interactions","author":"Miech","year":"2020"},{"key":"ref29","article-title":"UCF101: A dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"ref30","article-title":"A short note on the kinetics-700 human action dataset","author":"Carreira","year":"2019"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01891-x"},{"key":"ref34","article-title":"Tip-adapter: Training-free clip-adapter for better vision-language modeling","author":"Zhang","year":"2021"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01129"},{"key":"ref36","first-page":"1","article-title":"Open-vocabulary object detection via vision and language knowledge distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gu"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1023\/b:visi.0000029664.99615.94"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1023\/A:1010933404324"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01519"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01059"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25178"},{"key":"ref45","first-page":"9125","article-title":"DetCLIP: Dictionary-enriched visual-concept paralleled pre-training for open-world detection","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Yao"},{"key":"ref46","article-title":"SegCLIP: Patch aggregation with learnable centers for open-vocabulary semantic segmentation","author":"Luo","year":"2022"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1644"},{"key":"ref49","first-page":"1","article-title":"Deep fragment embeddings for bidirectional image sentence mapping","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"27","author":"Karpathy"},{"key":"ref50","article-title":"Vision-language intelligence: Tasks, representation learning, and large models","author":"Li","year":"2022"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/762"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-022-1369-5"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2023.3275156"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-022-1410-8"},{"key":"ref55","first-page":"1","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","volume":"28","author":"Ren"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"ref57","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref59","first-page":"6105","article-title":"EfficientNet: Rethinking model scaling for convolutional neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tan"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00065"},{"key":"ref61","first-page":"7324","article-title":"Making convolutional networks shift-invariant again","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref63","first-page":"12077","article-title":"SegFormer: Simple and efficient design for semantic segmentation with transformers","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Xie"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01857"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01553"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"ref68","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref69","first-page":"18661","article-title":"Supervised contrastive learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Khosla"},{"key":"ref70","article-title":"BEIT: BERT pre-training of image transformers","author":"Bao","year":"2021"},{"key":"ref71","first-page":"32942","article-title":"Coarse-to-fine vision-language pre-training with fusion in the backbone","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dou"},{"key":"ref72","article-title":"VLMO: Unified vision-language pre-training with mixture-of-modality-experts","author":"Bao","year":"2021"},{"key":"ref73","first-page":"1143","article-title":"Im2Text: Describing images using 1 million captioned photographs","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ordonez"},{"key":"ref74","article-title":"Microsoft COCO captions: Data collection and evaluation server","author":"Chen","year":"2015"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1145\/2812802"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_38"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463257"},{"key":"ref81","article-title":"RedCaps: Web-curated image-text data created by the people, for the people","author":"Desai","year":"2021"},{"key":"ref82","article-title":"Wukong: 100 million large-scale chinese cross-modal pre-training dataset and a foundation framework","author":"Gu","year":"2022"},{"key":"ref83","article-title":"PaLI: A jointly-scaled multilingual language-image model","author":"Chen","year":"2022"},{"key":"ref84","first-page":"1008","article-title":"UniCLIP: Unified framework for contrastive language-image pre-training","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Lee"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00852"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/759"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2004.383"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"ref92","first-page":"1","article-title":"Reading digits in natural images with unsupervised feature learning","author":"Netzer","year":"2011","journal-title":"Proc. Int. Conf. Neural Inf. Process. Syst. Workshop"},{"key":"ref93","first-page":"215","article-title":"An analysis of single-layer networks in unsupervised feature learning","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Coates"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2011.6033395"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.5244\/C.26.127"},{"key":"ref96","article-title":"Fine-grained visual classification of aircraft","author":"Maji","year":"2013"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-42051-1_16"},{"key":"ref98","first-page":"1631","article-title":"Recursive deep models for semantic compositionality over a sentiment treebank","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Socher"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.461"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.259"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2017.2675998"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-00934-2_24"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2019.2918242"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00550"},{"key":"ref108","article-title":"Elevater: A benchmark and toolkit for evaluating language-augmented visual models","author":"Li","year":"2022"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.119"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.544"},{"key":"ref112","first-page":"1","article-title":"Data efficient language-supervised zero-shot recognition with optimal transport distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wu"},{"key":"ref113","article-title":"Supervision exists everywhere: A data efficient contrastive language-image pre-training paradigm","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_14"},{"key":"ref115","article-title":"Florence: A new foundation model for computer vision","author":"Yuan","year":"2021"},{"key":"ref116","article-title":"Pyramidclip: Hierarchical feature alignment for vision-language model pretraining","author":"Gao","year":"2022"},{"key":"ref117","article-title":"Chinese clip: Contrastive vision-language pretraining in chinese","author":"Yang","year":"2022"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01759"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.552"},{"key":"ref120","article-title":"Large-scale bilingual language-image contrastive learning","author":"Ko","year":"2022"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01061"},{"key":"ref122","article-title":"K-lite: Learning transferable visual models with external knowledge","author":"Shen","year":"2022"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25172"},{"key":"ref124","article-title":"HiCLIP: Contrastive language-image pretraining with hierarchy-aware attention","author":"Geng","year":"2023"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01846"},{"key":"ref126","article-title":"Improving clip training with language rewrites","author":"Fan","year":"2023"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00273"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02027"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01760"},{"key":"ref130","article-title":"Perceptual grouping in vision-language models","author":"Ranasinghe","year":"2022"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.1109\/tcsvt.2023.3245584"},{"key":"ref133","article-title":"Language-aware soft prompting for vision & language foundation models","author":"Bulat","year":"2022"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00514"},{"key":"ref135","article-title":"Variational prompt tuning improves generalization of vision-language models","author":"Lu","year":"2022"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.01435"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.224"},{"key":"ref138","article-title":"Prompt learning with optimal transport for vision-language models","author":"Chen","year":"2022"},{"key":"ref139","first-page":"30569","article-title":"DualCoOp: Fast adaptation to multi-label recognition with limited annotations","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Sun"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00275"},{"key":"ref141","article-title":"Prompt tuning with soft context sharing for vision-language models","author":"Ding","year":"2022"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"ref143","article-title":"Unsupervised prompt learning for vision-language models","author":"Huang","year":"2022"},{"key":"ref144","first-page":"14274","article-title":"Test-time prompt tuning for zero-shot generalization in vision-language models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Shu"},{"key":"ref145","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"ref146","article-title":"Protect: Prompt tuning for hierarchical consistency","author":"Wu","year":"2023"},{"key":"ref147","article-title":"Exploring visual prompts for adapting large-scale models","author":"Bahng","year":"2022"},{"key":"ref148","article-title":"Retrieval-enhanced visual prompt learning for few-shot classification","author":"Rong","year":"2023"},{"key":"ref149","article-title":"Unified vision and language prompt learning","author":"Zang","year":"2022"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1109\/wacv57701.2024.00556"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01832"},{"key":"ref152","article-title":"Class-aware visual prompt tuning for vision-language pre-trained model","author":"Xing","year":"2022"},{"key":"ref153","article-title":"SVL-adapter: Self-supervised adapter for vision-language pretrained models","author":"Pantazis","year":"2022"},{"key":"ref154","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.00257"},{"key":"ref155","article-title":"Improving zero-shot models with label distribution priors","author":"Kahana","year":"2022"},{"key":"ref156","doi-asserted-by":"publisher","DOI":"10.1109\/tmm.2023.3311646"},{"key":"ref157","article-title":"VT-CLIP: Enhancing vision-language models with visual-guided texts","author":"Zhang","year":"2021"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25152"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01049"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.01438"},{"key":"ref161","article-title":"Visual classification via description from large language models","author":"Menon","year":"2022"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00780"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_40"},{"key":"ref164","article-title":"Masked unsupervised self-training for zero-shot image classification","author":"Li","year":"2022"},{"key":"ref165","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"ref167","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Houlsby"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02025"},{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00288"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01282"},{"key":"ref172","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01075"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01368"},{"key":"ref177","first-page":"33781","article-title":"Bridging the gap between object and image-level representations for open-vocabulary detection","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Rasheed"},{"key":"ref178","article-title":"ZSD-YOLO: Zero-shot YOLO detection using vision-language knowledgedistillation","author":"Xie","year":"2021"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01076"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01464"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01072"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_41"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_16"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00689"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2023.3293484"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19818-2_42"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"ref189","first-page":"1","article-title":"Language-driven semantic segmentation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li"},{"key":"ref190","article-title":"Semantic segmentation in-the-wild without seeing any segmentation examples","author":"Zabari","year":"2021"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01863"},{"key":"ref192","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01469"},{"key":"ref193","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00444"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00681"},{"key":"ref196","article-title":"Learning object-language alignments for open-vocabulary object detection","author":"Lin","year":"2022"},{"key":"ref197","article-title":"F-VLM: Open-vocabulary object detection upon frozen vision and language models","author":"Kuo","year":"2022"},{"key":"ref198","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"ref199","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_10"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00845"},{"key":"ref201","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00682"},{"key":"ref202","first-page":"33754","article-title":"RECO: Retrieve and co-segment for zero-shot transfer","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Shin"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/10582780\/10445007.pdf?arnumber=10445007","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T04:44:24Z","timestamp":1720241064000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10445007\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8]]},"references-count":202,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3369699","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8]]}}}