{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:19:24Z","timestamp":1782317964117,"version":"3.54.5"},"reference-count":93,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/access.2024.3512379","type":"journal-article","created":{"date-parts":[[2024,12,5]],"date-time":"2024-12-05T19:13:43Z","timestamp":1733426023000},"page":"187329-187342","source":"Crossref","is-referenced-by-count":2,"title":["Tuning-Free Universally-Supervised Semantic Segmentation"],"prefix":"10.1109","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-7885-302X","authenticated-orcid":false,"given":"Xiaobo","family":"Yang","sequence":"first","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9955-3569","authenticated-orcid":false,"given":"Xiaojin","family":"Gong","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref2","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. ICML","author":"Radford"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref4","article-title":"Dinov2: Learning robust visual features without supervision","volume-title":"arXiv:2304.07193","author":"Oquab","year":"2023"},{"key":"ref5","article-title":"Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection","volume-title":"arXiv:2303.05499","author":"Liu","year":"2023"},{"key":"ref6","article-title":"Segment anything model (SAM) enhanced pseudo labels for weakly supervised semantic segmentation","volume-title":"arXiv:2305.05803","author":"Chen","year":"2023"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00058"},{"key":"ref8","article-title":"An alternative to WSSS? An empirical study of the segment anything model (SAM) on weakly-supervised semantic segmentation problems","volume-title":"arXiv:2305.01586","author":"Sun","year":"2023"},{"key":"ref9","article-title":"Grounded sam: Assembling open-world models for diverse visual tasks","volume-title":"arXiv:2401.14159","author":"Ren","year":"2024"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01619"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_40"},{"key":"ref12","first-page":"1","article-title":"Learning with local and global consistency","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"16","author":"Zhou"},{"key":"ref13","article-title":"SCLIP: Rethinking self-attention for dense vision-language inference","volume-title":"arXiv:2312.01597","author":"Wang","year":"2023"},{"key":"ref14","article-title":"Grounding everything: Emerging localization properties in vision-language transformers","volume-title":"arXiv:2312.00878","author":"Bousselham","year":"2023"},{"key":"ref15","article-title":"A closer look at the explainability of contrastive language-image pre-training","volume-title":"arXiv:2304.05653","author":"Li","year":"2023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00143"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3418210"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00080"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00682"},{"key":"ref20","article-title":"CLIPSelf: Vision transformer distills itself for open-vocabulary dense prediction","volume-title":"arXiv:2310.01403","author":"Wu","year":"2023"},{"key":"ref21","first-page":"23033","article-title":"Segclip: Patch aggregation with learnable centers for open-vocabulary semantic segmentation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Luo"},{"key":"ref22","article-title":"Open-vocabulary segmentation with semantic-assisted calibration","volume-title":"arXiv:2312.04089","author":"Liu","year":"2023"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00288"},{"key":"ref24","article-title":"Open-vocabulary universal image segmentation with MaskCLIP","volume-title":"arXiv:2208.08984","author":"Ding","year":"2022"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00106"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00367"},{"key":"ref27","article-title":"Open-vocabulary SAM: Segment and recognize twenty-thousand classes interactively","volume-title":"arXiv:2401.02955","author":"Yuan","year":"2024"},{"key":"ref28","article-title":"Open-vocabulary segmentation with unpaired mask-text supervision","volume-title":"arXiv:2402.08960","author":"Wang","year":"2024"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02684"},{"key":"ref30","article-title":"PosSAM: Panoptic open-vocabulary segment anything","volume-title":"arXiv:2403.09620","author":"VS","year":"2024"},{"key":"ref31","first-page":"35631","article-title":"Learning mask-aware clip representations for zero-shot segmentation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Jiao"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00521"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00297"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00353"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02190"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2007.190672"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1802.02611"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"ref39","article-title":"PseudoSeg: Designing pseudo labels for semantic segmentation","volume-title":"arXiv:2010.09713","author":"Zou","year":"2020"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00718"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01484"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01092"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00304"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00699"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01498"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00119"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01484"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611906"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00348"},{"key":"ref50","article-title":"The faiss library","volume-title":"arXiv:2401.08281","author":"Douze","year":"2024"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126343"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00132"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00431"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6971"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01634"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01590-z"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20056-4_26"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00302"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3350176"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00339"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00346"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72992-8_26"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2966647"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00425"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00444"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01877"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02267"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01875"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01469"},{"key":"ref73","article-title":"Question-answer cross language image matching for weakly supervised semantic segmentation","volume-title":"arXiv:2401.09883","author":"Deng","year":"2024"},{"key":"ref74","article-title":"Segment anything is a good pseudo-label generator forweakly supervised semantic segmentation","volume-title":"arXiv:2305.015861","author":"Jiang","year":"2023"},{"key":"ref75","article-title":"WeakTr: Exploring plain vision transformer for weakly-supervised semantic segmentation","volume-title":"arXiv:2304.01184","author":"Zhu","year":"2023"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00320"},{"key":"ref77","first-page":"1","article-title":"Efficient inference in fully connected CRFs with Gaussian edge potentials","volume-title":"Proc. NIPS","author":"Philipp"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01760"},{"key":"ref79","first-page":"33754","article-title":"Reco: Retrieve and co-segment for zero-shot transfer","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Shin"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01074"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00149"},{"key":"ref82","first-page":"1","article-title":"An image is worth 16\u00d716 words: Transformers for image recognition at scale","volume-title":"Proc. ICLR","author":"Dosovitskiy"},{"key":"ref83","article-title":"Masked autoencoders are scalable vision learners","volume-title":"arXiv:2111.06377","author":"He","year":"2021"},{"key":"ref84","first-page":"1","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Paszke"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00523"},{"key":"ref89","volume-title":"Accelerating Generative AI With PyTorch: Segment Anything, Fast\u2014PyTorch","year":"2024"},{"key":"ref90","article-title":"Faster segment anything: Towards lightweight SAM for mobile applications","volume-title":"arXiv:2306.14289","author":"Zhang","year":"2023"},{"key":"ref91","article-title":"Fast segment anything","volume-title":"arXiv:2306.12156","author":"Zhao","year":"2023"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00427"},{"key":"ref93","article-title":"MobileCLIP: Fast image-text models through multi-modal reinforced training","volume-title":"arXiv:2311.17049","author":"Kumar Anasosalu Vasu","year":"2023"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10380310\/10779462.pdf?arnumber=10779462","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,18]],"date-time":"2024-12-18T19:40:45Z","timestamp":1734550845000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10779462\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":93,"URL":"https:\/\/doi.org\/10.1109\/access.2024.3512379","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}