{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:17:58Z","timestamp":1781587078406,"version":"3.54.5"},"reference-count":89,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Science Fund of China","award":["62361166670"],"award-info":[{"award-number":["62361166670"]}]},{"name":"National Science Fund of China","award":["U24A20330"],"award-info":[{"award-number":["U24A20330"]}]},{"name":"National Science Fund of China","award":["62206134"],"award-info":[{"award-number":["62206134"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["070-63233084"],"award-info":[{"award-number":["070-63233084"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1109\/tpami.2024.3504568","type":"journal-article","created":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T19:10:14Z","timestamp":1732216214000},"page":"1594-1609","source":"Crossref","is-referenced-by-count":11,"title":["Fine-Grained Visual Text Prompting"],"prefix":"10.1109","volume":"47","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2725-8947","authenticated-orcid":false,"given":"Lingfeng","family":"Yang","sequence":"first","affiliation":[{"name":"PCA Lab, Key Lab of Intelligent Perception and Systems for High-Dimensional Information of Ministry of Education, and Jiangsu Key Lab of Image and Video Understanding for Social Security, School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4996-7365","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"additional","affiliation":[{"name":"NKIARI, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yueze","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing Academy of Artificial Intelligence (BAAI), Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6974-7976","authenticated-orcid":false,"given":"Xinlong","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing Academy of Artificial Intelligence (BAAI), Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4800-832X","authenticated-orcid":false,"given":"Jian","family":"Yang","sequence":"additional","affiliation":[{"name":"PCA Lab, Key Lab of Intelligent Perception and Systems for High-Dimensional Information of Ministry of Education, and Jiangsu Key Lab of Image and Video Understanding for Social Security, School of Computer Science and Engineering, Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Evaluating clip: Towards characterization of broader capabilities and downstream implications","author":"Agarwal","year":"2021"},{"key":"ref2","article-title":"Flamingo: A visual language model for few-shot learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Alayrac"},{"key":"ref3","article-title":"Exploring visual prompts for adapting large-scale models","author":"Bahng","year":"2022"},{"key":"ref4","first-page":"33781","article-title":"Bridging the gap between object and image-level representations for open-vocabulary detection","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Bangalath"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19784-0_41"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01764"},{"key":"ref7","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref8","first-page":"213","article-title":"End-to-end object detection with transformers","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Carion"},{"key":"ref9","article-title":"Pix2seq: A language modeling framework for object detection","author":"Chen","year":"2021"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.254"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"ref12","first-page":"17864","article-title":"Per-pixel classification is not all you need for semantic segmentation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Cheng"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"ref14","article-title":"Palm: Scaling language modeling with pathways","author":"Chowdhery","year":"2022"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.343"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"ref17","first-page":"9358","article-title":"EVA: Exploring the limits of masked visual representation learning at scale","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","author":"Fang"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"ref19","first-page":"770","article-title":"Instance-level human parsing via part grouping network","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Gong"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.715"},{"key":"ref21","article-title":"Open-vocabulary object detection via vision and language knowledge distillation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gu"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01362"},{"key":"ref23","first-page":"2961","article-title":"Mask R-CNN","volume-title":"Proc. IEEE Int. Conf. Comput. Vis.","author":"He"},{"key":"ref24","first-page":"709","article-title":"Visual prompt tuning","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Jia"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01507"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00159"},{"key":"ref28","first-page":"5583","article-title":"ViLT: Vision-and-language transformer without convolution or region supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00963"},{"key":"ref30","doi-asserted-by":"crossref","DOI":"10.1109\/ICCV51070.2023.00371","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"key":"ref31","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1002\/nav.3800020109","article-title":"The Hungarian method for the assignment problem","volume":"2","author":"Kuhn","year":"1955","journal-title":"Nav. Res. logistics Quart."},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206594"},{"key":"ref33","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"ref34","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref35","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref36","first-page":"10955","article-title":"Grounded language-image pre-training","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","author":"Harold"},{"key":"ref37","first-page":"121","article-title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Li"},{"key":"ref38","article-title":"Scaling language-image pre-training via masking","author":"Li","year":"2022"},{"key":"ref39","article-title":"Open-vocabulary semantic segmentation with mask-adapted clip","author":"Liang","year":"2022"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00477"},{"key":"ref42","article-title":"Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection","author":"Liu","year":"2023"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00270"},{"key":"ref44","first-page":"539","volume-title":"Proc. ACM Int. Conf. Multimedia","author":"Liu"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.9"},{"key":"ref47","first-page":"529","article-title":"SLIP: Self-supervision meets language-image pre-training","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Mu"},{"key":"ref48","article-title":"When does label smoothing help?","author":"M\u00fcller","year":"2019"},{"key":"ref49","article-title":"GPT-4 technical report","year":"2023"},{"key":"ref50","article-title":"Kosmos-2: Grounding multimodal large language models to the world","author":"Peng","year":"2023"},{"key":"ref51","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref52","first-page":"140:1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00690"},{"key":"ref54","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref55","first-page":"8821","article-title":"Zero-shot text-to-image generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ramesh"},{"key":"ref56","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ren"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.346"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.01101"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.357"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3058684"},{"key":"ref61","article-title":"Eva-clip: Improved training techniques for clip at scale","author":"Sun","year":"2023"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01251"},{"key":"ref63","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-030-34372-9","volume-title":"Computer Vision: Algorithms and Applications","author":"Szeliski","year":"2022"},{"key":"ref64","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref65","article-title":"The Caltech-UCSD birds-200\u20132011 dataset","author":"Wah","year":"2011"},{"key":"ref66","article-title":"Large language model is also an open-ended decoder for vision-centric tasks","author":"Wang","year":"2023"},{"key":"ref67","first-page":"649","article-title":"SOLO: Segmenting objects by locations","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Wang"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01378"},{"key":"ref69","article-title":"SegGPT: Segmenting everything in context","author":"Wang","year":"2023"},{"key":"ref70","article-title":"PyTorch image models","author":"Wightman","year":"2019"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00366"},{"key":"ref72","first-page":"648","article-title":"Zoom better to see clearer: Human and object parsing with hierarchical auto-zoom net","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Xia"},{"key":"ref73","first-page":"13723","article-title":"Tune-an-ellipse: Clip has potential to find what you want","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","author":"Xie"},{"key":"ref74","doi-asserted-by":"crossref","DOI":"10.1109\/CVPR52729.2023.01471","article-title":"Universal instance perception as object discovery and retrieval","author":"Yan","year":"2023"},{"key":"ref75","first-page":"24993","article-title":"Fine-grained visual prompting","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref76","first-page":"387","article-title":"Improving one-stage visual grounding by recursive sub-query construction","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Yang"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00478"},{"key":"ref78","article-title":"CPT: Colorful prompt tuning for pre-trained vision-language models","author":"Yao","year":"2021"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00142"},{"key":"ref80","first-page":"69","article-title":"Modeling context in referring expressions","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Yu"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01864"},{"key":"ref82","first-page":"106","article-title":"Open-vocabulary DETR with conditional matching","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Zang"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00437"},{"key":"ref84","first-page":"834","article-title":"Part-based R-CNNs for fine-grained category detection","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Zhang"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.660"},{"key":"ref87","first-page":"159","article-title":"Exploiting unlabeled data with vision and language models for object detection","volume-title":"Proc. Eur. Conf. Comput. Vis.","author":"Zhao"},{"key":"ref88","first-page":"16816","article-title":"Conditional prompt learning for vision-language models","volume-title":"Proc. IEEE Conf. Comp. Vis. Patt. Recogn.","author":"Zhou"},{"key":"ref89","doi-asserted-by":"crossref","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","article-title":"Learning to prompt for vision-language models","volume":"130","author":"Zhou","year":"2022","journal-title":"Int. J. Comput. Vis."}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10873290\/10763465.pdf?arnumber=10763465","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,29]],"date-time":"2025-03-29T05:55:17Z","timestamp":1743227717000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10763465\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3]]},"references-count":89,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3504568","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3]]}}}