{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T03:28:31Z","timestamp":1777865311809,"version":"3.51.4"},"reference-count":62,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012240","name":"Academic Excellence Foundation of BUAA for PhD Students","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012240","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.01943","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"20900-20910","source":"Crossref","is-referenced-by-count":0,"title":["Visual Textualization for Image Prompted Object Detection"],"prefix":"10.1109","author":[{"given":"Yongjian","family":"Wu","sequence":"first","affiliation":[{"name":"School of Biological Science and Medical Engineering, Beihang University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Biological Science and Medical Engineering, Beihang University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiya","family":"Saiyin","sequence":"additional","affiliation":[{"name":"School of Biological Science and Medical Engineering, Beihang University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bingzheng","family":"Wei","sequence":"additional","affiliation":[{"name":"ByteDance Inc."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Biological Science and Medical Engineering, Beihang University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1148\/radiol.2323032035"},{"key":"ref2","article-title":"Exploring visual prompts for adapting largescale models","author":"Bahng","year":"2022","journal-title":"arXiv preprint"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01083"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.629"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2387"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.01737"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"ref8","author":"Efrat","journal-title":"arXiv preprint"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr42600.2020.00407"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87237-3_13"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2019.101563"},{"key":"ref13","article-title":"A systematic survey of prompt engineering on vision-language foundation models","author":"Gu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref14","article-title":"Openvocabulary object detection via vision and language knowledge distillation","author":"Gu","year":"2021","journal-title":"arXiv preprint"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.111964"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00550"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02703"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00325"},{"key":"ref19","article-title":"Multi-modal fewshot object detection with meta-learning-based cross-modal prompting","author":"Han","year":"2022","journal-title":"arXiv preprint"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19959"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00525"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00525"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25153"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"ref25","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of naacL-HLT, page 2","author":"Devlin","year":"2019"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.01832"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2017.2677499"},{"key":"ref28","article-title":"F-vlm: Open-vocabulary object detection upon frozen vision and language models","author":"Kuo","year":"2022","journal-title":"arXiv preprint"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0675"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25216"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01069"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25274"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00313"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3191"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00856"},{"issue":"8","key":"ref41","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"ref42","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford","year":"2021"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72667-5_17"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00852"},{"key":"ref45","first-page":"7173","article-title":"Fewshot adaptive faster r-cnn","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","author":"Wang","year":"2019"},{"key":"ref46","article-title":"Frustratingly simple few-shot object detection","author":"Wang","journal-title":"arXiv preprint"},{"key":"ref47","article-title":"Frustratingly simple few-shot object detection","author":"Wang","journal-title":"arXiv preprint"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01192"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_27"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_34"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00679"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01888"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.23977\/acss.2023.070812"},{"key":"ref54","article-title":"Deeplesion: Automated deep mining, categorization and detection of significant radiology image findings using large-scale clinical lesion annotations","author":"Yan","year":"2017","journal-title":"arXiv preprint"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00967"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3195735"},{"key":"ref57","article-title":"Detect every thing with few examples","author":"Zhang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.124926"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01584"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445179.pdf?arnumber=11445179","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T06:14:04Z","timestamp":1777529644000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445179\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":62,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.01943","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}