{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T18:05:42Z","timestamp":1779386742362,"version":"3.53.1"},"reference-count":38,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T00:00:00Z","timestamp":1772150400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.patcog.2026.113393","type":"journal-article","created":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T22:49:19Z","timestamp":1772405359000},"page":"113393","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Feature space sharing via visual projector for open-vocabulary multi-label classification"],"prefix":"10.1016","volume":"178","author":[{"given":"Zhanfang","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaohan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"7","key":"10.1016\/j.patcog.2026.113393_bib0001","doi-asserted-by":"crossref","first-page":"1425","DOI":"10.1109\/TPAMI.2015.2487986","article-title":"Label-embedding for image classification","volume":"38","author":"Akata","year":"2016","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell"},{"key":"10.1016\/j.patcog.2026.113393_bib0002","doi-asserted-by":"crossref","first-page":"6027","DOI":"10.1007\/s00371-024-03769-6","article-title":"Open-vocabulary multi-label classification with visual and textual features fusion","volume":"41","author":"Liu","year":"2025","journal-title":"Vis. Comput"},{"key":"10.1016\/j.patcog.2026.113393_bib0003","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"6034","article-title":"Zero-shot learning via joint latent similarity embedding","author":"Zhang","year":"2016"},{"key":"10.1016\/j.patcog.2026.113393_bib0004","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"8773","article-title":"A shared multi-attention framework for multi-label zero-shot learning","author":"Huynh","year":"2020"},{"key":"10.1016\/j.patcog.2026.113393_bib0005","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"620","article-title":"Semantic diversity learning for zero-shot multi-label classification","author":"Ben-Cohen","year":"2021"},{"key":"10.1016\/j.patcog.2026.113393_bib0006","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"8711","article-title":"Discriminative region-based multi-label zero-shot learning","author":"Narayan","year":"2021"},{"key":"10.1016\/j.patcog.2026.113393_bib0007","series-title":"Proc. Conf. Empirical Methods Nat. Lang. Process. (EMNLP)","first-page":"1532","article-title":"GloVe: global vectors for word representation","author":"Pennington","year":"2014"},{"key":"10.1016\/j.patcog.2026.113393_bib0008","first-page":"52","article-title":"An empirical study and analysis of generalized zero-shot learning for object recognition in the wild","volume":"9906","author":"Chao","year":"2016"},{"key":"10.1016\/j.patcog.2026.113393_bib0009","first-page":"808","article-title":"Open-vocabulary multi-label classification via multi-modal knowledge transfer","volume":"37","author":"He","year":"2023","journal-title":"Proc. AAAI Conf. Artif. Intell"},{"issue":"1","key":"10.1016\/j.patcog.2026.113393_bib0010","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1007\/s11633-022-1369-5","article-title":"VLP: a survey on vision-language pre-training","volume":"20","author":"Chen","year":"2023","journal-title":"Mach. Intell. Res"},{"key":"10.1016\/j.patcog.2026.113393_bib0011","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"139","author":"Radford","year":"2021","journal-title":"Proc. Int. Conf. Mach. Learn. (ICML)"},{"issue":"9","key":"10.1016\/j.patcog.2026.113393_bib0012","doi-asserted-by":"crossref","first-page":"2251","DOI":"10.1109\/TPAMI.2018.2857768","article-title":"Zero-shot learning\u2014A comprehensive evaluation of the good, the bad and the ugly","volume":"41","author":"Xian","year":"2019","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell"},{"key":"10.1016\/j.patcog.2026.113393_bib0013","series-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"1348","article-title":"CDUL: cLIP-driven unsupervised learning for multi-label image classification","author":"Abdelfattah","year":"2023"},{"key":"10.1016\/j.patcog.2026.113393_bib0014","first-page":"32897","article-title":"VLMo: unified vision-language pre-training with mixture-of-modality-experts","volume":"35","author":"Bao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst"},{"key":"10.1016\/j.patcog.2026.113393_bib0015","doi-asserted-by":"crossref","first-page":"438","DOI":"10.1109\/ITS.1998.718433","article-title":"Adaptive channel equalization using neural network","volume":"2","author":"Albu","year":"1998","journal-title":"Proc. SBT\/IEEE Int. Telecommun. Symp. (ITS)"},{"key":"10.1016\/j.patcog.2026.113393_bib0016","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110648","article-title":"Prompt-guided DETR with RoI-pruned masked attention for open-vocabulary object detection","volume":"155","author":"Song","year":"2024","journal-title":"Pattern Recognit"},{"issue":"9","key":"10.1016\/j.patcog.2026.113393_bib0017","first-page":"1735","article-title":"Transductive multi-label zero-shot learning","volume":"39","author":"Fu","year":"2017","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell"},{"key":"10.1016\/j.patcog.2026.113393_bib0018","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"951","article-title":"Learning to detect unseen object classes by between-class attribute transfer","author":"Lampert","year":"2009"},{"key":"10.1016\/j.patcog.2026.113393_bib0019","series-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","article-title":"Zero-shot learning by convex combination of semantic embeddings","author":"Norouzi","year":"2014"},{"key":"10.1016\/j.patcog.2026.113393_bib0020","series-title":"Proc. ACM SIGIR Conf. Res. Dev. Inf. Retr.","first-page":"879","article-title":"Zero-shot image tagging by hierarchical semantic embedding","author":"Li","year":"2015"},{"key":"10.1016\/j.patcog.2026.113393_bib0021","series-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"5985","article-title":"Fast zero-shot image tagging","author":"Zhang","year":"2016"},{"issue":"12","key":"10.1016\/j.patcog.2026.113393_bib0022","doi-asserted-by":"crossref","first-page":"14611","DOI":"10.1109\/TPAMI.2023.3295772","article-title":"Generative multi-label zero-shot learning","volume":"45","author":"Gupta","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell"},{"key":"10.1016\/j.patcog.2026.113393_bib0023","series-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","article-title":"VisualBERT: a simple and performant baseline for vision and language","author":"Li","year":"2020"},{"key":"10.1016\/j.patcog.2026.113393_bib0024","first-page":"9694","article-title":"Align before fuse: vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst"},{"key":"10.1016\/j.patcog.2026.113393_bib0025","series-title":"Proc. Int. Conf. Mach. Learn. (ICML)","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume":"139","author":"Jia","year":"2021"},{"key":"10.1016\/j.patcog.2026.113393_bib0026","first-page":"728","article-title":"Simple open-vocabulary object detection with vision transformers","volume":"13670","author":"Zhou","year":"2022"},{"issue":"9","key":"10.1016\/j.patcog.2026.113393_bib0027","doi-asserted-by":"crossref","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","article-title":"Learning to prompt for vision-language models","volume":"130","author":"Zhou","year":"2022","journal-title":"Int. J. Comput. Vis"},{"issue":"7","key":"10.1016\/j.patcog.2026.113393_bib0028","first-page":"38","article-title":"Distilling the knowledge in a neural network","volume":"14","author":"Hinton","year":"2015","journal-title":"Comput. Sci."},{"key":"10.1016\/j.patcog.2026.113393_bib0029","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)","first-page":"4582","article-title":"Prefix-tuning: optimizing continuous prompts for generation","author":"Li","year":"2021"},{"issue":"8","key":"10.1016\/j.patcog.2026.113393_bib0030","doi-asserted-by":"crossref","first-page":"1819","DOI":"10.1109\/TKDE.2013.39","article-title":"A review on multi-label learning algorithms","volume":"26","author":"Zhang","year":"2014","journal-title":"IEEE Trans. Knowl. Data Eng"},{"key":"10.1016\/j.patcog.2026.113393_bib0031","article-title":"DeViSE: a deep visual-semantic embedding model","volume":"26","author":"Frome","year":"2013","journal-title":"Adv. Neural Inf. Process. Syst"},{"issue":"12","key":"10.1016\/j.patcog.2026.113393_bib0032","doi-asserted-by":"crossref","DOI":"10.1145\/3762195","article-title":"Query-based knowledge sharing for open-vocabulary multi-label classification","volume":"21","author":"Zhu","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl"},{"key":"10.1016\/j.patcog.2026.113393_bib0033","series-title":"Proc. ACM Int. Conf. Image Video Retr. (CIVR)","first-page":"1","article-title":"NUS-WIDE: a real-world web image database from National University of Singapore","author":"Chua","year":"2009"},{"key":"10.1016\/j.patcog.2026.113393_bib0034","series-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"},{"key":"10.1016\/j.patcog.2026.113393_bib0035","first-page":"30569","article-title":"DualCoOp: fast adaptation to multi-label recognition with limited annotations","volume":"35","author":"Sun","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst"},{"key":"10.1016\/j.patcog.2026.113393_bib0036","series-title":"Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"23859","article-title":"(ML)\u00b2P-Encoder: on exploration of channel-class correlation for multi-label zero-shot learning","author":"Liu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113393_bib0037","series-title":"Proc. AAAI Conf. Artif. Intell.","first-page":"3513","article-title":"TagCLIP: a local-to-global framework to enhance open-vocabulary multi-label classification of CLIP without training","author":"Lin","year":"2024"},{"issue":"1","key":"10.1016\/j.patcog.2026.113393_bib0038","doi-asserted-by":"crossref","first-page":"242","DOI":"10.1109\/TMM.2019.2924511","article-title":"Deep0tag: deep multiple instance learning for zero-shot image tagging","volume":"22","author":"Rahman","year":"2019","journal-title":"IEEE Trans. Multimed."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003584?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003584?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T17:04:56Z","timestamp":1779383096000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326003584"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":38,"alternative-id":["S0031320326003584"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113393","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Feature space sharing via visual projector for open-vocabulary multi-label classification","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113393","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113393"}}