{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:53:40Z","timestamp":1784300020187,"version":"3.55.0"},"reference-count":60,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00893","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"9572-9582","source":"Crossref","is-referenced-by-count":1,"title":["Teaching VLMs to Localize Specific Objects from In-Context Examples"],"prefix":"10.1109","author":[{"given":"Sivan","family":"Doveh","sequence":"first","affiliation":[{"name":"Weizmann Institute of Science"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nimrod","family":"Shabtay","sequence":"additional","affiliation":[{"name":"IBM Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eli","family":"Schwartz","sequence":"additional","affiliation":[{"name":"IBM Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hilde","family":"Kuehne","sequence":"additional","affiliation":[{"name":"IBM Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raja","family":"Giryes","sequence":"additional","affiliation":[{"name":"Tel Aviv University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rogerio","family":"Feris","sequence":"additional","affiliation":[{"name":"MIT-IBM"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Leonid","family":"Karlinsky","sequence":"additional","affiliation":[{"name":"MIT-IBM"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"James","family":"Glass","sequence":"additional","affiliation":[{"name":"MIT CSAIL"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Assaf","family":"Arbelle","sequence":"additional","affiliation":[{"name":"IBM Research"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shimon","family":"Ullman","sequence":"additional","affiliation":[{"name":"Weizmann Institute of Science"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"M. Jehanzeb","family":"Mirza","sequence":"additional","affiliation":[{"name":"MIT CSAIL"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Pixtral 12b","author":"Agrawal","year":"2024","journal-title":"arXiv preprint"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref3","article-title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond","volume":"2","author":"Bai","year":"2023","journal-title":"arXiv preprint"},{"key":"ref4","article-title":"Decimamba: Exploring the length extrapolation potential of mamba","author":"Ben-Kish","year":"2024","journal-title":"arXiv preprint"},{"key":"ref5","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref6","article-title":"Language Models are Few-Shot Learners","author":"Brown","year":"2020","journal-title":"NeurIPS"},{"key":"ref7","article-title":"MiniGPT-v2: Large Language Model as a Unified Interface for Vision-Language Multi-task Learning","volume-title":"Proc. ICLR","author":"Chen","year":"2024"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4231-5"},{"key":"ref9","author":"Chiang","year":"2023","journal-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2142"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_26"},{"key":"ref12","article-title":"Dense and Aligned Captions (DAC) Promote Compositional Reasoning in VL Models","author":"Doveh","year":"2023","journal-title":"NeurIPS"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00261"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-93806-1_19"},{"key":"ref15","article-title":"The Llama 3 Herd of Models","author":"Dubey","year":"2024","journal-title":"arXiv preprint"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00552"},{"key":"ref17","article-title":"SEED: Self-supervised Distillation for Visual Representation","volume-title":"Proc. ICLR","volume":"6","author":"Fang","year":"2021"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73636-0_15"},{"key":"ref19","article-title":"Task vectors are crossmodal","author":"Bar","year":"2024","journal-title":"arXiv preprint"},{"key":"ref20","article-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2021","journal-title":"arXiv preprint"},{"key":"ref21","article-title":"Multimodal task vectors enable many-shot multimodal in-context learning","volume":"2","author":"Huang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref22","article-title":"ConMe: Rethinking Evaluation of Compositional Reasoning for Modern VLMs","author":"Huang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2957464"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00686"},{"key":"ref25","article-title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","volume-title":"Proc. ICML","author":"Jia","year":"2021"},{"key":"ref26","author":"Kahana","year":"2022","journal-title":"Improving Zero-Shot Models with Label Distribution Priors"},{"issue":"3","key":"ref27","volume":"2","author":"Lauren\u00e7on","year":"2024","journal-title":"Building and better understanding vision-language models: insights and future directions."},{"key":"ref28","author":"Lauren\u00e7on","year":"2023","journal-title":"Obelics: An open webscale filtered dataset of interleaved image-text documents"},{"issue":"6","key":"ref29","article-title":"LLaVA- OneVision: Easy Visual Task Transfer","volume":"3","author":"Li","year":"2024","journal-title":"arXiv preprint"},{"key":"ref30","article-title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","volume-title":"In Proc. ICML","author":"Li","year":"2023"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00267"},{"key":"ref35","author":"Liu","year":"2023","journal-title":"LLaVA- NeXT: Improved reasoning, OCR, and world knowledge"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.201"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72627-9_21"},{"key":"ref41","article-title":"TAP: Targeted Prompting for Task Adaptive Generation of Textual Training Instances for Visual Classification","author":"Jehanzeb","year":"2023","journal-title":"arXiv preprint"},{"key":"ref42","article-title":"LaFTer: Label-Free Tuning of Zero-shot Classifier using Language and Unlabeled Image Collections","author":"Jehanzeb Mirza","year":"2023","journal-title":"NeurIPS"},{"key":"ref43","article-title":"Glov: Guided large language models as implicit optimizers for vision language models","author":"Mirza","year":"2024","journal-title":"arXiv preprint"},{"key":"ref44","article-title":"GPT-4 Technical Report","volume":"2","year":"2023","journal-title":"arXiv preprint"},{"key":"ref45","article-title":"Train short, test long: Attention with linear biases enables input length extrapolation","author":"Press","year":"2021","journal-title":"arXiv preprint"},{"key":"ref46","article-title":"Learning Transferable Visual Models from Natural Language Supervision","volume-title":"Proc. ICML","volume":"2","author":"Radford","year":"2021"},{"key":"ref47","article-title":"Where\u2019s waldo: Diffusion features for personalized segmentation and retrieval","author":"Samuel","year":"2024","journal-title":"NeurIPS"},{"key":"ref48","article-title":"LAION-5b: An open large-scale dataset for training next generation image-text models","author":"Schuhmann","year":"2022","journal-title":"NeurIPS"},{"key":"ref49","article-title":"Generative multimodal models are in-context learners","author":"Sun","year":"2023","journal-title":"arXiv preprint"},{"issue":"3","key":"ref50","article-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","volume":"2","author":"Wang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref51","article-title":"Frustratingly simple few-shot object detection","author":"Wang","year":"2020","journal-title":"arXiv preprint"},{"key":"ref52","article-title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","volume":"2","author":"Wei","year":"2022","journal-title":"NeurIPS"},{"key":"ref53","article-title":"Larger language models do in-context learning differently","author":"Wei","year":"2023","journal-title":"arXiv preprint"},{"key":"ref54","article-title":"Demystifying CLIP Data","volume-title":"Proc. ICLR","author":"Xu","year":"2023"},{"key":"ref55","article-title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","volume":"2","author":"Yao","year":"2023","journal-title":"In NeurIPS"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"ref57","article-title":"Personalize segment anything model with one shot","author":"Zhang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref58","article-title":"Mmicl: Empowering vision-language model with multi-modal in-context learning","author":"Zhao","year":"2023","journal-title":"arXiv preprint"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-demos.38"},{"key":"ref60","article-title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","volume-title":"Proc. ICLR","author":"Zhu","year":"2024"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11444257.pdf?arnumber=11444257","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T06:46:05Z","timestamp":1777531565000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11444257\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":60,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00893","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}