{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T15:50:47Z","timestamp":1778860247688,"version":"3.51.4"},"publisher-location":"Cham","reference-count":49,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729454","type":"print"},{"value":"9783031729461","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T00:00:00Z","timestamp":1727827200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T00:00:00Z","timestamp":1727827200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72946-1_21","type":"book-chapter","created":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T19:02:08Z","timestamp":1727809328000},"page":"368-385","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Parrot Captions Teach CLIP to\u00a0Spot Text"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8208-3705","authenticated-orcid":false,"given":"Yiqi","family":"Lin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8697-695X","authenticated-orcid":false,"given":"Conghui","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alex Jinpeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5625-2966","authenticated-orcid":false,"given":"Bin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weijia","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,2]]},"reference":[{"key":"21_CR1","unstructured":"Agarwal, S., Krueger, G., Clark, J., Radford, A., Kim, J.W., Brundage, M.: Evaluating CLIP: towards characterization of broader capabilities and downstream implications. arXiv preprint arXiv:2108.02818 (2021)"},{"key":"21_CR2","unstructured":"Alabdulmohsin, I., Wang, X., Steiner, A.P., Goyal, P., D\u2019Amour, A., Zhai, X.: CLIP the bias: how useful is balancing data in multimodal learning? In: ICLR (2024)"},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Ali, J., Kleindessner, M., Wenzel, F., Budhathoki, K., Cevher, V., Russell, C.: Evaluating the fairness of discriminative foundation models in computer vision. In: AIES, pp. 809\u2013833 (2023)","DOI":"10.1145\/3600211.3604720"},{"key":"21_CR4","doi-asserted-by":"crossref","unstructured":"Berg, H., et al.: A prompt array keeps the bias away: debiasing vision-language models with adversarial learning. arXiv preprint arXiv:2203.11933 (2022)","DOI":"10.18653\/v1\/2022.aacl-main.61"},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Biten, A.F., et al.: Scene text visual question answering. In: ICCV, pp. 4291\u20134301 (2019)","DOI":"10.1109\/ICCV.2019.00439"},{"key":"21_CR6","unstructured":"Byeon, M., Park, B., Kim, H., Lee, S., Baek, W., Kim, S.: COYO-700M: image-text pair dataset (2022). https:\/\/github.com\/kakaobrain\/coyo-dataset"},{"key":"21_CR7","unstructured":"Cao, L., et al.: Less is more: removing text-regions improves CLIP training efficiency and robustness. arXiv preprint arXiv:2305.05095 (2023)"},{"key":"21_CR8","unstructured":"Chen, X., et al.: Microsoft COCO captions: data collection and evaluation server. arXiv preprint arXiv:1504.00325 (2015)"},{"key":"21_CR9","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"21_CR10","unstructured":"Dosovitskiy, A., et al.: An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"21_CR11","unstructured":"Gadre, S.Y., et al.: DataComp: in search of the next generation of multimodal datasets. arXiv preprint arXiv:2304.14108 (2023)"},{"key":"21_CR12","doi-asserted-by":"crossref","unstructured":"Ganz, R., Nuriel, O., Aberdam, A., Kittenplon, Y., Mazor, S., Litman, R.: Towards models that can see and read. arXiv preprint arXiv:2301.07389 (2023)","DOI":"10.1109\/ICCV51070.2023.01985"},{"issue":"3","key":"21_CR13","doi-asserted-by":"publisher","DOI":"10.23915\/distill.00030","volume":"6","author":"G Goh","year":"2021","unstructured":"Goh, G., et al.: Multimodal neurons in artificial neural networks. Distill 6(3), e30 (2021)","journal-title":"Distill"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: CVPR, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"21_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"21_CR16","unstructured":"Ilharco, G., et al.: OpenCLIP (2021). https:\/\/doi.org\/10.5281\/zenodo.5143773"},{"key":"21_CR17","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: ICML, pp. 4904\u20134916. PMLR (2021)"},{"issue":"3","key":"21_CR18","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1109\/TBDATA.2019.2921572","volume":"7","author":"J Johnson","year":"2019","unstructured":"Johnson, J., Douze, M., J\u00e9gou, H.: Billion-scale similarity search with GPUs. IEEE Trans. Big Data 7(3), 535\u2013547 (2019)","journal-title":"IEEE Trans. Big Data"},{"key":"21_CR19","unstructured":"Lemesle, Y., Sawayama, M., Valle-Perez, G., Adolphe, M., Sauz\u00e9on, H., Oudeyer, P.Y.: Language-biased image classification: evaluation based on semantic representations. arXiv preprint arXiv:2201.11014 (2022)"},{"key":"21_CR20","unstructured":"Li, B., Weinberger, K.Q., Belongie, S., Koltun, V., Ranftl, R.: Language-driven semantic segmentation. In: ICLR (2022)"},{"key":"21_CR21","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML, pp. 12888\u201312900. PMLR (2022)"},{"key":"21_CR22","unstructured":"Li, J., Selvaraju, R., Gotmare, A., Joty, S., Xiong, C., Hoi, S.C.H.: Align before fuse: vision and language representation learning with momentum distillation. In: NeurIPS, vol. 34, pp. 9694\u20139705 (2021)"},{"key":"21_CR23","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"21_CR24","unstructured":"Maini, P., Goyal, S., Lipton, Z.C., Kolter, J.Z., Raghunathan, A.: T-MARS: improving visual representations by circumventing text feature learning. arXiv preprint arXiv:2307.03132 (2023)"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Materzy\u0144ska, J., Torralba, A., Bau, D.: Disentangling visual and written concepts in CLIP. In: CVPR, pp. 16410\u201316419 (2022)","DOI":"10.1109\/CVPR52688.2022.01592"},{"key":"21_CR26","unstructured":"Nichol, A., et al.: GLIDE: towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741 (2021)"},{"key":"21_CR27","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126658","volume":"555","author":"H Pham","year":"2023","unstructured":"Pham, H., et al.: Combined scaling for zero-shot transfer learning. Neurocomputing 555, 126658 (2023)","journal-title":"Neurocomputing"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Radenovic, F., et al.: Filtering, distillation, and hard negatives for vision-language pre-training. In: CVPR, pp. 6967\u20136977 (2023)","DOI":"10.1109\/CVPR52729.2023.00673"},{"key":"21_CR29","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763. PMLR (2021)"},{"key":"21_CR30","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"21_CR31","unstructured":"Schuhmann, C., et al.: LAION-5B: an open large-scale dataset for training next generation image-text models. In: NeurIPS, vol. 35, pp. 25278\u201325294 (2022)"},{"key":"21_CR32","unstructured":"Schuhmann, C., et al.: LAION-400M: open dataset of CLIP-filtered 400 million image-text pairs. arXiv preprint arXiv:2111.02114 (2021)"},{"key":"21_CR33","doi-asserted-by":"crossref","unstructured":"Shi, C., Yang, S.: LoGoPrompt: synthetic text images can be good visual prompts for vision-language models. In: ICCV, pp. 2932\u20132941 (2023)","DOI":"10.1109\/ICCV51070.2023.00274"},{"key":"21_CR34","doi-asserted-by":"crossref","unstructured":"Shtedritski, A., Rupprecht, C., Vedaldi, A.: What does CLIP know about a red circle? Visual prompt engineering for VLMs. arXiv preprint arXiv:2304.06712 (2023)","DOI":"10.1109\/ICCV51070.2023.01101"},{"key":"21_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"742","DOI":"10.1007\/978-3-030-58536-5_44","volume-title":"Computer Vision \u2013 ECCV 2020","author":"O Sidorov","year":"2020","unstructured":"Sidorov, O., Hu, R., Rohrbach, M., Singh, A.: TextCaps: a dataset for image captioning with reading comprehension. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12347, pp. 742\u2013758. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58536-5_44"},{"key":"21_CR36","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards VQA models that can read. In: CVPR, pp. 8317\u20138326 (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"21_CR37","doi-asserted-by":"crossref","unstructured":"Tanjim, M.M., Singh, K.K., Kafle, K., Sinha, R., Cottrell, G.W.: Discovering and mitigating biases in CLIP-based image editing. In: WACV, pp. 2984\u20132993 (2024)","DOI":"10.1109\/WACV57701.2024.00296"},{"issue":"1","key":"21_CR38","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1080\/10867651.2004.10487596","volume":"9","author":"A Telea","year":"2004","unstructured":"Telea, A.: An image inpainting technique based on the fast marching method. J. Graph. Tools 9(1), 23\u201334 (2004)","journal-title":"J. Graph. Tools"},{"key":"21_CR39","doi-asserted-by":"crossref","unstructured":"Tschannen, M., Mustafa, B., Houlsby, N.: CLIPPO: image-and-language understanding from pixels only. In: CVPR, pp. 11006\u201311017 (2023)","DOI":"10.1109\/CVPR52729.2023.01059"},{"key":"21_CR40","unstructured":"Wang, H., Zhan, Y., Liu, L., Ding, L., Yu, J.: Balanced similarity with auxiliary prompts: towards alleviating text-to-image retrieval bias for CLIP in zero-shot learning. arXiv preprint arXiv:2402.18400 (2024)"},{"key":"21_CR41","doi-asserted-by":"crossref","unstructured":"Wang, J., Liu, Y., Wang, X.E.: Are gender-neutral queries really gender-neutral? Mitigating gender bias in image search. arXiv preprint arXiv:2109.05433 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.151"},{"key":"21_CR42","unstructured":"Wei, J., et al.: Emergent abilities of large language models. arXiv preprint arXiv:2206.07682 (2022)"},{"key":"21_CR43","unstructured":"Xu, Y., Xu, Z., Chai, W., Zhao, Z., Song, E., Wang, G.: Devil in the number: towards robust multi-modality data filter. arXiv preprint arXiv:2309.13770 (2023)"},{"key":"21_CR44","unstructured":"Yao, Y., Zhang, A., Zhang, Z., Liu, Z., Chua, T.S., Sun, M.: CPT: colorful prompt tuning for pre-trained vision-language models. arXiv preprint arXiv:2109.11797 (2021)"},{"key":"21_CR45","doi-asserted-by":"crossref","unstructured":"Ye, M., et al.: DeepSolo: let transformer decoder with explicit points solo for text spotting. In: CVPR, pp. 19348\u201319357 (2023)","DOI":"10.1109\/CVPR52729.2023.01854"},{"key":"21_CR46","unstructured":"Yuksekgonul, M., Bianchi, F., Kalluri, P., Jurafsky, D., Zou, J.: When and why vision-language models behave like bags-of-words, and what to do about it? In: ICLR (2022)"},{"key":"21_CR47","doi-asserted-by":"crossref","unstructured":"Zhai, X., et al.: LiT: zero-shot transfer with locked-image text tuning. In: CVPR, pp. 18123\u201318133 (2022)","DOI":"10.1109\/CVPR52688.2022.01759"},{"key":"21_CR48","doi-asserted-by":"publisher","first-page":"1141","DOI":"10.1007\/s11263-022-01739-w","volume":"131","author":"Q Zhang","year":"2023","unstructured":"Zhang, Q., Xu, Y., Zhang, J., Tao, D.: ViTAEv2: vision transformer advanced by exploring inductive bias for image recognition and beyond. IJCV 131, 1141\u20131162 (2023)","journal-title":"IJCV"},{"issue":"9","key":"21_CR49","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. IJCV 130(9), 2337\u20132348 (2022)","journal-title":"IJCV"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72946-1_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T23:35:55Z","timestamp":1732836955000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72946-1_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,2]]},"ISBN":["9783031729454","9783031729461"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72946-1_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,2]]},"assertion":[{"value":"2 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}