{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,7]],"date-time":"2026-02-07T12:55:30Z","timestamp":1770468930482,"version":"3.49.0"},"publisher-location":"Cham","reference-count":80,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031736605","type":"print"},{"value":"9783031736612","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,10]],"date-time":"2024-11-10T00:00:00Z","timestamp":1731196800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,10]],"date-time":"2024-11-10T00:00:00Z","timestamp":1731196800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73661-2_14","type":"book-chapter","created":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T11:06:33Z","timestamp":1731150393000},"page":"247-265","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["HYPE: Hyperbolic Entailment Filtering for\u00a0Underspecified Images and\u00a0Texts"],"prefix":"10.1007","author":[{"given":"Wonjae","family":"Kim","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sanghyuk","family":"Chun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Taekyung","family":"Kim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongyoon","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sangdoo","family":"Yun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,10]]},"reference":[{"key":"14_CR1","doi-asserted-by":"crossref","unstructured":"Atigh, M.G., Schoep, J., Acar, E., Van Noord, N., Mettes, P.: Hyperbolic image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4453\u20134462 (2022)","DOI":"10.1109\/CVPR52688.2022.00441"},{"key":"14_CR2","unstructured":"Bai, Y., Ying, Z., Ren, H., Leskovec, J.: Modeling heterogeneous hierarchies with relation-specific hyperbolic cones. In: Advance in Neural Information Processing System, vol. 34, pp. 12316\u201312327 (2021)"},{"key":"14_CR3","unstructured":"Haim, R.B., et al .: The second PASCAL recognising textual entailment challenge (2006)"},{"key":"14_CR4","unstructured":"Barbu, A., et al.: Objectnet: a large-scale bias-controlled dataset for pushing the limits of object recognition models. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"14_CR5","unstructured":"Bentivogli, L., Clark, P., Dagan, I., Giampiccolo, D.: The fifth PASCAL recognizing textual entailment challenge (2009)"},{"key":"14_CR6","unstructured":"Bitton, Y., et al.: Gamified association benchmark to challenge vision-and-language models. In: Advance in Neural Information Processing System, vol. 35, pp. 26549\u201326564 (2022)"},{"key":"14_CR7","unstructured":"Byeon, M., Park, B., Kim, H., Lee, S., Baek, W., Kim, S.: Coyo-700m: image-text pair dataset (2022). https:\/\/github.com\/kakaobrain\/coyo-dataset"},{"key":"14_CR8","doi-asserted-by":"crossref","unstructured":"Changpinyo, S., Sharma, P., Ding, N., Soricut, R.: Conceptual 12M: pushing web-scale image-text pre-training to recognize long-tail visual concepts. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"14_CR9","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"14_CR10","doi-asserted-by":"crossref","unstructured":"Chen, X., Xie, S., He, K.: An empirical study of training self-supervised vision transformers. arXiv preprint arXiv:2104.02057 (2021)","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"14_CR11","doi-asserted-by":"crossref","unstructured":"Cherti, M., et al.: Reproducible scaling laws for contrastive language-image learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132829 (2023)","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"14_CR12","unstructured":"Chun, S.: Improved probabilistic image-text representations. In: International Conference on Learning Representations (2024)"},{"key":"14_CR13","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-031-20074-8_1","volume-title":"ECCV 2022","author":"S Chun","year":"2022","unstructured":"Chun, S., Kim, W., Park, S., Chang, M., Oh, S.J.: ECCV caption: correcting false negatives by collecting machine-and-human-verified image-caption associations for MS-COCO. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13688, pp. 1\u201319. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20074-8_1"},{"key":"14_CR14","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1007\/11736790_9","volume-title":"Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment","author":"I Dagan","year":"2006","unstructured":"Dagan, I., Glickman, O., Magnini, B.: The PASCAL recognising textual entailment challenge. In: Qui\u00f1onero-Candela, J., Dagan, I., Magnini, B., d\u2019Alch\u00e9-Buc, F. (eds.) MLCW 2005. LNCS (LNAI), vol. 3944, pp. 177\u2013190. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11736790_9"},{"key":"14_CR15","unstructured":"Desai, K., Kaul, G., Aysola, Z., Johnson, J.: RedCaps: web-curated image-text data created by the people, for the people. In: NeurIPS Datasets and Benchmarks (2021)"},{"key":"14_CR16","unstructured":"Desai, K., Nickel, M., Rajpurohit, T., Johnson, J., Vedantam, R.: Hyperbolic image-text representations. In: Proceedings of the International Conference on Machine Learning (2023)"},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Dhall, A., Makarova, A., Ganea, O., Pavllo, D., Greeff, M., Krause, A.: Hierarchical image classification using entailment cone embeddings. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 836\u2013837 (2020)","DOI":"10.1109\/CVPRW50498.2020.00426"},{"key":"14_CR18","doi-asserted-by":"crossref","unstructured":"Ermolov, A., Mirvakhabova, L., Khrulkov, V., Sebe, N., Oseledets, I.: Hyperbolic vision transformers: combining improvements in metric learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7409\u20137419 (2022)","DOI":"10.1109\/CVPR52688.2022.00726"},{"key":"14_CR19","unstructured":"Fang, A., Jose, A.M., Jain, A., Schmidt, L., Toshev, A., Shankar, V.: Data filtering networks. arXiv preprint arXiv:2309.17425 (2023)"},{"key":"14_CR20","unstructured":"Gadre, S.Y., et al.: Datacomp: in search of the next generation of multimodal datasets. arXiv preprint arXiv:2304.14108 (2023)"},{"key":"14_CR21","unstructured":"Ganea, O., B\u00e9cigneul, G., Hofmann, T.: Hyperbolic entailment cones for learning hierarchical embeddings. In: International Conference on Machine Learning, pp. 1646\u20131655. PMLR (2018)"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Ge, S., Mishra, S., Kornblith, S., Li, C-L., Jacobs, D.: Hyperbolic contrastive learning for visual representations beyond objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6840\u20136849 (2023)","DOI":"10.1109\/CVPR52729.2023.00661"},{"key":"14_CR23","unstructured":"Geng, X., Liu, H., Lee, L., Schuurmans, D., Levine, S., Abbeel, P.: Multimodal masked autoencoders learn transferable representations. arXiv preprint arXiv:2205.14204 (2022)"},{"key":"14_CR24","doi-asserted-by":"crossref","unstructured":"Giampiccolo, D., Magnini, B., Dagan, I., Dolan, B.: The third PASCAL recognizing textual entailment challenge. In: Proceedings of the ACL-PASCAL Workshop on Textual Entailment and Paraphrasing, pp. 1\u20139. Association for Computational Linguistics (2007)","DOI":"10.3115\/1654536.1654538"},{"key":"14_CR25","unstructured":"Gutmann, M., Hyv\u00e4rinen, A.: Noise-contrastive estimation: a new estimation principle for unnormalized statistical models. In: Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics, pp. 297\u2013304. JMLR Workshop and Conference Proceedings (2010)"},{"key":"14_CR26","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., et al.: The many faces of robustness: a critical analysis of out-of-distribution generalization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8340\u20138349 (2021)","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"14_CR27","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., Song, D.: Natural adversarial examples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15262\u201315271 (2021)","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"14_CR28","unstructured":"Hernandez, D., Kaplan, J., Henighan, T., McCandlish, S.: Scaling laws for transfer. arXiv preprint arXiv:2102.01293 (2021)"},{"key":"14_CR29","unstructured":"Hestness, J., et al.: Deep learning scaling is predictable, empirically. arXiv preprint arXiv:1712.00409 (2017)"},{"key":"14_CR30","unstructured":"Hoffmann, J., et al.: An empirical analysis of compute-optimal large language model training. In: Advances in Neural Information Processing Systems, vol. 35, pp. 30016\u201330030 (2022)"},{"key":"14_CR31","unstructured":"Huang, T-H., Shin, C., Tay, S.J., Adila, D., Sala, F.: Multimodal data curation via object detection and filter ensembles (2023)"},{"key":"14_CR32","unstructured":"Ilharco, G., et al.: Openclip, 2021. If you use this software, please cite it as below"},{"key":"14_CR33","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning, pp. 4904\u20134916. PMLR (2021)"},{"key":"14_CR34","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"14_CR35","doi-asserted-by":"crossref","unstructured":"Khrulkov, V., Mirvakhabova, L., Ustinova, E., Oseledets, I., Lempitsky, V.: Hyperbolic image embeddings. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6418\u20136428 (2020)","DOI":"10.1109\/CVPR42600.2020.00645"},{"key":"14_CR36","unstructured":"Koh, P.W., et al.: Wilds: a benchmark of in-the-wild distribution shifts. In: International Conference on Machine Learning, pp. 5637\u20135664. PMLR (2021)"},{"key":"14_CR37","unstructured":"Koppula, S., et al.: Where should i spend my flops? efficiency evaluations of visual pre-training methods. arXiv preprint arXiv:2209.15589 (2022)"},{"key":"14_CR38","doi-asserted-by":"crossref","unstructured":"Le, M., Roller, S., Papaxanthos, L., Kiela, D., Nickel, M.: Inferring concept hierarchies from text corpora via hyperbolic embeddings. arXiv preprint arXiv:1902.00913 (2019)","DOI":"10.18653\/v1\/P19-1313"},{"key":"14_CR39","unstructured":"LeCun, Y.: The mnist database of handwritten digits (1998). http:\/\/yann.lecun.com\/exdb\/mnist\/"},{"key":"14_CR40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-91755-9","volume-title":"Introduction to Riemannian Manifolds","author":"JM Lee","year":"2018","unstructured":"Lee, J.M.: Introduction to Riemannian Manifolds. GTM, Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-319-91755-9"},{"key":"14_CR41","unstructured":"Levesque, H.J., Davis, E., Morgenstern, L.: The Winograd schema challenge. In: AAAI Spring Symposium: Logical Formalizations of Commonsense Reasoning, p. 47 (2011)"},{"key":"14_CR42","unstructured":"Li, L., et\u00a0al.: Value: a multi-task benchmark for video-and-language understanding evaluation. arXiv preprint arXiv:2106.04632 (2021)"},{"key":"14_CR43","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"14_CR44","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2017)"},{"key":"14_CR45","unstructured":"Maini, P., Goyal, S., Lipton, Z.C., Kolter, J.Z., Raghunathan, A.: T-mars: improving visual representations by circumventing text feature learning. arXiv preprint arXiv:2307.03132 (2023)"},{"key":"14_CR46","volume-title":"Raum und Zeit","author":"H Minkowski","year":"1988","unstructured":"Minkowski, H.: Raum und Zeit. Springer, Cham (1988)"},{"key":"14_CR47","unstructured":"Netzer, Y., Wang, T., Coates, A., Bissacco, A., Wu, B., Ng, A.Y.: Reading digits in natural images with unsupervised feature learning (2011)"},{"key":"14_CR48","unstructured":"Nguyen, T., Ilharco, G., Wortsman, M., Sewoong, O., Schmidt, L.: Quality not quantity: on the interaction between dataset design and robustness of clip. In: Advance in Neural Information Processing System, vol. 35, pp. 21455\u201321469 (2022)"},{"key":"14_CR49","unstructured":"Nickel, M., Kiela, D.: Poincar\u00e9 embeddings for learning hierarchical representations. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"14_CR50","unstructured":"Oord, A., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"14_CR51","doi-asserted-by":"publisher","first-page":"126658","DOI":"10.1016\/j.neucom.2023.126658","volume":"555","author":"H Pham","year":"2023","unstructured":"Pham, H., et al.: Combined scaling for zero-shot transfer learning. Neurocomputing 555, 126658 (2023)","journal-title":"Neurocomputing"},{"key":"14_CR52","unstructured":"Purushwalkam, S., Gupta, A.: Demystifying contrastive self-supervised learning: Invariances, augmentations and dataset biases. In: Advance in Neural Information Processing System, vol. 33, pp. 3407\u20133418 (2020)"},{"key":"14_CR53","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference in Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"14_CR54","doi-asserted-by":"crossref","unstructured":"Ranasinghe, K., McKinzie, B., Ravi, S., Yang, Y., Toshev, A., Shlens, J.: Perceptual grouping in contrastive vision-language models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5571\u20135584 (2023)","DOI":"10.1109\/ICCV51070.2023.00513"},{"key":"14_CR55","unstructured":"Recht, B., Roelofs, R., Schmidt, L., Shankar, V.: Do imagenet classifiers generalize to ImageNet? In: Proceedings of the 36th International Conference on Machine Learning, pp. 5389\u20135400. PMLR (2019)"},{"key":"14_CR56","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: Imagenet large scale visual recognition challenge. IJCV 115, 211\u2013252 (2015)","journal-title":"IJCV"},{"key":"14_CR57","unstructured":"Schuhmann, C., et al.: Laion-400m: open dataset of clip-filtered 400 million image-text pairs. arXiv preprint arXiv:2111.02114 (2021)"},{"key":"14_CR58","unstructured":"Schuhmann, C., et al.: Laion-5b: an open large-scale dataset for training next generation image-text models. In: Advance in Neural Information Processing System, vol. 35, pp. 25278\u201325294 (2022)"},{"key":"14_CR59","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of ACL (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"14_CR60","doi-asserted-by":"crossref","unstructured":"Shon, S., et al.: New benchmark tasks for spoken language understanding evaluation on natural speech. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7927\u20137931. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746137"},{"key":"14_CR61","unstructured":"Sorscher, B., Geirhos, R., Shekhar, S., Ganguli, S., Morcos, A.: Beyond neural scaling laws: beating power law scaling via data pruning. In: Advance in Neural Information Processing System , vol. 35, pp. 19523\u201319536 (2022)"},{"key":"14_CR62","doi-asserted-by":"crossref","unstructured":"Sun, C., Shrivastava, A., Singh, S., Gupta, A.: Revisiting unreasonable effectiveness of data in deep learning era. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2017)","DOI":"10.1109\/ICCV.2017.97"},{"key":"14_CR63","unstructured":"Tifrea, A., B\u00e9cigneul, G., Ganea, O-E.: Poincar$$\\backslash $$\u2019e glove: hyperbolic word embeddings (2019)"},{"key":"14_CR64","unstructured":"Valencia, V.M.S.: Studies on natural logic and categorial grammar. Universiteit van Amsterdam (1991)"},{"key":"14_CR65","unstructured":"Valentino, M., Carvalho, D.S., Freitas, A.: Multi-relational hyperbolic word embeddings from natural language definitions (2023)"},{"key":"14_CR66","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-009-4540-1","volume-title":"Essays in Logical Semantics","author":"J Van Benthem","year":"1986","unstructured":"Van Benthem, J., et al.: Essays in Logical Semantics. Springer, Cham (1986)"},{"key":"14_CR67","unstructured":"Vendrov, I., Kiros, R., Fidler, S., Urtasun, R.: Order-embeddings of images and language. arXiv preprint arXiv:1511.06361 (2015)"},{"key":"14_CR68","doi-asserted-by":"crossref","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., Bowman, S.: GLUE: a multi-task benchmark and analysis platform for natural language understanding. In: Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, pp. 353\u2013355. Association for Computational Linguistics Brussels, Belgium (2018)","DOI":"10.18653\/v1\/W18-5446"},{"key":"14_CR69","unstructured":"Wang, A., et al.: Superglue: a stickier benchmark for general-purpose language understanding systems. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"14_CR70","doi-asserted-by":"crossref","unstructured":"Williams, A., Nangia, N., Bowman, S.: A broad-coverage challenge corpus for sentence understanding through inference. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers), pp. 1112\u20131122. Association for Computational Linguistics (2018)","DOI":"10.18653\/v1\/N18-1101"},{"key":"14_CR71","unstructured":"Xie, N., Lai, F., Doran, D., Kadav, A.: Visual entailment task for visually-grounded language learning. arXiv preprint arXiv:1811.10582 (2018)"},{"key":"14_CR72","unstructured":"Xie, N., Lai, F., Doran, D., Kadav, A.: Visual entailment: a novel task for fine-grained image understanding. arXiv preprint arXiv:1901.06706 (2019)"},{"key":"14_CR73","unstructured":"Xu, H., et al.: Demystifying clip data. arXiv preprint arXiv:2309.16671 (2023)"},{"key":"14_CR74","unstructured":"Yalniz, I.Z., J\u00e9gou, H., Chen, K., Paluri, M., Mahajan, D.: Billion-scale semi-supervised learning for image classification (2019)"},{"key":"14_CR75","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguist. 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"14_CR76","unstructured":"Yu, H., Tian, Y., Kumar, S., Yang, L., Wang, H.: The devil is in the details: a deep dive into the rabbit hole of data filtering. arXiv preprint arXiv:2309.15954 (2023)"},{"key":"14_CR77","unstructured":"Yue, Y., Lin, F., Yamada, K.D., Zhang, Z.: Hyperbolic contrastive learning (2023)"},{"key":"14_CR78","unstructured":"Zhai, X., et al.: A large-scale study of representation learning with the visual task adaptation benchmark. arXiv preprint arXiv:1910.04867 (2019)"},{"key":"14_CR79","doi-asserted-by":"crossref","unstructured":"Zhai, X., Kolesnikov, A., Houlsby, N., Beyer, L.: Scaling vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12104\u201312113 (2022)","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"14_CR80","unstructured":"Zhou, W., Zeng, Y., Diao, S., Zhang, X.: Vlue: a multi-task multi-dimension benchmark for evaluating vision-language pre-training. In: International Conference on Machine Learning, pp. 27395\u201327411. PMLR (2022)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73661-2_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T12:05:13Z","timestamp":1731153913000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73661-2_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,10]]},"ISBN":["9783031736605","9783031736612"],"references-count":80,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73661-2_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,10]]},"assertion":[{"value":"10 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}