{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T04:48:38Z","timestamp":1782103718279,"version":"3.54.5"},"reference-count":64,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134289","type":"journal-article","created":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T16:29:33Z","timestamp":1781713773000},"page":"134289","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MiddleCLIP: Unleashing the potential of the middle layers in CLIP for training-free open-vocabulary semantic segmentation"],"prefix":"10.1016","volume":"698","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5130-9505","authenticated-orcid":false,"given":"Xiao","family":"Jin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dai-Wei","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3361-4207","authenticated-orcid":false,"given":"Wei","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7824-0985","authenticated-orcid":false,"given":"Zhuang","family":"Shao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134289_bib0005","series-title":"Proceedings of the International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2026.134289_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129982","article-title":"GCD-net: global consciousness-driven open-vocabulary semantic segmentation network","volume":"636","author":"Wu","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0015","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129702","article-title":"Image\u2013text aggregation for open-vocabulary semantic segmentation","volume":"630","author":"Cheng","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.132229","article-title":"Efficient redundancy reduction for open-vocabulary semantic segmentation","volume":"665","author":"Chen","year":"2026","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.131844","article-title":"FA-Seg: a fast and accurate diffusion-based method for open-vocabulary segmentation","volume":"660","author":"Che","year":"2026","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.128788","article-title":"Physically-guided open vocabulary segmentation with weighted patched alignment loss","volume":"614","author":"Liu","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0035","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2026.133088","article-title":"SPSRL: open-vocabulary semantic segmentation with spatial prior and semantic relation learning","volume":"677","author":"Ping","year":"2026","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134289_bib0040","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"14824","article-title":"DeCLIP: decoupled learning for open-vocabulary dense perception","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0045","series-title":"Proceedings of the IEEE Computer Vision and Pattern Recognition Conference","first-page":"29968","article-title":"ResCLIP: residual attention for training-free dense vision-language inference","author":"Yang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0050","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"11165","article-title":"Learning to generate text-grounded mask for open-world semantic segmentation from only image-text pairs","author":"Cha","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0055","series-title":"Proceedings of the International Conference on Machine Learning","first-page":"23033","article-title":"SegCLIP: patch aggregation with learnable centers for open-vocabulary semantic segmentation","author":"Luo","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0060","doi-asserted-by":"crossref","first-page":"68798","DOI":"10.52202\/075280-3011","article-title":"Rewrite caption semantics: bridging semantic gaps for language-supervised semantic segmentation","volume":"36","author":"Xing","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134289_bib0065","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"18134","article-title":"GroupViT: semantic segmentation emerges from text supervision","author":"Xu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134289_bib0070","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"2935","article-title":"Learning open-vocabulary semantic segmentation models from natural language supervision","author":"Xu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0075","doi-asserted-by":"crossref","first-page":"73652","DOI":"10.52202\/075280-3222","article-title":"Uncovering prototypical knowledge for weakly open-vocabulary semantic segmentation","volume":"36","author":"Zhang","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134289_bib0080","series-title":"Proceedings of the European Conference on Computer Vision","first-page":"315","article-title":"SCLIP: rethinking self-attention for dense vision-language inference","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0085","series-title":"Proceedings of the IEEE Winter Conference on Applications of Computer Vision","first-page":"5061","article-title":"Pay attention to your neighbours: training-free open-vocabulary semantic segmentation","author":"Hajimiri","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0090","series-title":"Proceedings of the European Conference on Computer Vision","first-page":"143","article-title":"ClearCLIP: decomposing clip representations for dense vision-language inference","author":"Lan","year":"2024"},{"issue":"1","key":"10.1016\/j.neucom.2026.134289_bib0095","doi-asserted-by":"crossref","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","article-title":"The pascal visual object classes challenge: a retrospective","volume":"111","author":"Everingham","year":"2015","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134289_bib0100","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"5649","article-title":"Unveiling the knowledge of CLIP for training-free open-vocabulary semantic segmentation","volume":"vol. 39","author":"Liu","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0105","series-title":"Proceedings of the International Conference on Machine Learning","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","author":"Jia","year":"2021"},{"key":"10.1016\/j.neucom.2026.134289_bib0110","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"2818","article-title":"Reproducible scaling laws for contrastive language-image learning","author":"Cherti","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0115","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3431","article-title":"Fully convolutional networks for semantic segmentation","author":"Long","year":"2015"},{"key":"10.1016\/j.neucom.2026.134289_bib0120","first-page":"12077","article-title":"SegFormer: simple and efficient design for semantic segmentation with transformers","volume":"34","author":"Xie","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134289_bib0125","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"15305","article-title":"CLIPis also an efficient segmenter: a text-driven approach for weakly supervised semantic segmentation","author":"Lin","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0130","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"5571","article-title":"Perceptual grouping in contrastive vision-language models","author":"Ranasinghe","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0135","doi-asserted-by":"crossref","first-page":"33754","DOI":"10.52202\/068431-2446","article-title":"Reco: retrieve and co-segment for zero-shot transfer","volume":"35","author":"Shin","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134289_bib0140","author":"Ren"},{"key":"10.1016\/j.neucom.2026.134289_bib0145","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"1290","article-title":"Masked-attention mask transformer for universal image segmentation","author":"Cheng","year":"2022"},{"key":"10.1016\/j.neucom.2026.134289_bib0150","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"7061","article-title":"Open-vocabulary semantic segmentation with mask-adapted clip","author":"Liang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0155","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3635","article-title":"SAM-CLIP: merging vision foundation models towards semantic and spatial understanding","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0160","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"2945","article-title":"Side adapter network for open-vocabulary semantic segmentation","author":"Xu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0165","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4113","article-title":"CAT-Seg: cost aggregation for open-vocabulary semantic segmentation","author":"Cho","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0170","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"13171","article-title":"Clip as RNN: segment countless visual concepts without training endeavor","author":"Sun","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0175","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4029","article-title":"Emergent open-vocabulary semantic segmentation from off-the-shelf vision-language models","author":"Luo","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0180","series-title":"Proceedings of the European Conference on Computer Vision","first-page":"696","article-title":"Extract free dense labels from clip","author":"Zhou","year":"2022"},{"key":"10.1016\/j.neucom.2026.134289_bib0185","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111409","article-title":"A closer look at the explainability of contrastive language-image pre-training","volume":"162","author":"Li","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.neucom.2026.134289_bib0190","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3828","article-title":"Grounding everything: emerging localization properties in vision-language transformers","author":"Bousselham","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0195","series-title":"Proceedings of the European Conference on Computer Vision","first-page":"70","article-title":"ProxyCLIP: proxy attention improves clip for open-vocabulary segmentation","author":"Lan","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0200","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"10545","article-title":"Segearth-ov: towards training-free open-vocabulary segmentation for remote sensing images","author":"Li","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0205","author":"Zhang"},{"issue":"12","key":"10.1016\/j.neucom.2026.134289_bib0210","doi-asserted-by":"crossref","first-page":"12038","DOI":"10.1109\/TCSVT.2025.3586282","article-title":"Cross-domain hyperspectral image classification based on bi-directional domain adaptation","volume":"35","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.neucom.2026.134289_bib0215","author":"Zhang"},{"key":"10.1016\/j.neucom.2026.134289_bib0220","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"23199","article-title":"Cliper: hierarchically improving spatial representation of clip for open-vocabulary semantic segmentation","author":"Sun","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0225","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"9794","article-title":"LPOSS: label propagation over patches and pixels for open-vocabulary semantic segmentation","author":"Stojni\u0107","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0230","series-title":"Proceedings of International Conference on Learning Representations","article-title":"Class distribution-induced attention map for open-vocabulary semantic segmentations","author":"Kang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0235","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"22815","article-title":"Plug-in feedback self-adaptive attention in clip for training-free open-vocabulary segmentation","author":"Chi","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0240","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"23124","article-title":"Training-free class purification for open-vocabulary semantic segmentation","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0245","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"891","article-title":"The role of context for object detection and semantic segmentation in the wild","author":"Mottaghi","year":"2014"},{"key":"10.1016\/j.neucom.2026.134289_bib0250","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"1209","article-title":"COCO-stuff: thing and stuff classes in context","author":"Caesar","year":"2018"},{"key":"10.1016\/j.neucom.2026.134289_bib0255","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3213","article-title":"The cityscapes dataset for semantic urban scene understanding","author":"Cordts","year":"2016"},{"issue":"3","key":"10.1016\/j.neucom.2026.134289_bib0260","doi-asserted-by":"crossref","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","article-title":"Semantic understanding of scenes through the ADE20k dataset","volume":"127","author":"Zhou","year":"2019","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134289_bib0265","series-title":"Proceedings of the IEEE Winter Conference on Applications of Computer Vision","first-page":"1464","article-title":"Fossil: free open-vocabulary semantic segmentation through synthetic references retrieval","author":"Barsellotti","year":"2024"},{"key":"10.1016\/j.neucom.2026.134289_bib0270","author":"Dosovitskiy"},{"key":"10.1016\/j.neucom.2026.134289_bib0275","unstructured":"M. Contributors, MMSegmentation: openmmlab semantic segmentation toolbox and benchmark, 2020."},{"key":"10.1016\/j.neucom.2026.134289_bib0280","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4253","article-title":"Single-stage semantic segmentation from image labels","author":"Araslanov","year":"2020"},{"key":"10.1016\/j.neucom.2026.134289_bib0285","article-title":"Efficient inference in fully connected CRFS with Gaussian edge potentials","volume":"24","author":"Kr\u00e4henb\u00fchl","year":"2011","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134289_bib0290","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"23487","article-title":"Harnessing vision foundation models for high-performance, training-free open vocabulary segmentation","author":"Shi","year":"2025"},{"key":"10.1016\/j.neucom.2026.134289_bib0295","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"6254","article-title":"Openearthmap: a benchmark dataset for global high-resolution land cover mapping","author":"Xia","year":"2023"},{"key":"10.1016\/j.neucom.2026.134289_bib0300","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134289_bib0305","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops","first-page":"28","article-title":"isaid: a large-scale dataset for instance segmentation in aerial images","author":"Waqas Zamir","year":"2019"},{"key":"10.1016\/j.neucom.2026.134289_bib0310","doi-asserted-by":"crossref","first-page":"108","DOI":"10.1016\/j.isprsjprs.2020.05.009","article-title":"UAVid: a semantic segmentation dataset for UAV imagery","volume":"165","author":"Lyu","year":"2020","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.neucom.2026.134289_bib0315","series-title":"Chinese Conference on Pattern Recognition and Computer Vision (PRCV)","first-page":"347","article-title":"Large-scale structure from motion with semantic constraints of aerial images","author":"Chen","year":"2018"},{"key":"10.1016\/j.neucom.2026.134289_bib0320","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2025.104429","article-title":"Vdd: varied drone dataset for semantic segmentation","volume":"109","author":"Cai","year":"2025","journal-title":"J. Vis. Commun. Image Represent."}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226016875?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226016875?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T04:13:37Z","timestamp":1782101617000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226016875"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":64,"alternative-id":["S0925231226016875"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134289","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MiddleCLIP: Unleashing the potential of the middle layers in CLIP for training-free open-vocabulary semantic segmentation","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134289","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"134289"}}