{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,25]],"date-time":"2026-02-25T17:52:07Z","timestamp":1772041927379,"version":"3.50.1"},"publisher-location":"Cham","reference-count":49,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031916717","type":"print"},{"value":"9783031916724","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91672-4_5","type":"book-chapter","created":{"date-parts":[[2025,5,20]],"date-time":"2025-05-20T15:25:00Z","timestamp":1747754700000},"page":"62-79","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Open-Vocabulary Object Detectors: Robustness Challenges Under Distribution Shifts"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6903-7552","authenticated-orcid":false,"given":"Prakash Chandra","family":"Chhipa","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0221-8268","authenticated-orcid":false,"given":"Kanjar","family":"De","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2770-6271","authenticated-orcid":false,"given":"Meenakshi Subhash","family":"Chippa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8532-0895","authenticated-orcid":false,"given":"Rajkumar","family":"Saini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4029-6574","authenticated-orcid":false,"given":"Marcus","family":"Liwicki","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"5_CR1","unstructured":"Achiam, J., et\u00a0al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"5_CR2","unstructured":"Arandjelovi\u0107, R., Andonian, A., Mensch, A., H\u00e9naff, O.J., Alayrac, J.B., Zisserman, A.: Three ways to improve feature alignment for open vocabulary detection. arXiv preprint arXiv:2303.13518 (2023)"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Bravo, M.A., Mittal, S., Brox, T.: Localized vision-language matching for open-vocabulary object detection. In: DAGM German Conference on Pattern Recognition, pp. 393\u2013408. Springer (2022)","DOI":"10.1007\/978-3-031-16788-1_24"},{"key":"5_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"5_CR5","unstructured":"Chen, P., et al.: Open vocabulary object detection with proposal mining and prediction equalization. arXiv preprint arXiv:2206.11134 (2022)"},{"key":"5_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., Shan, Y.: Yolo-world: real-time open-vocabulary object detection. In: Conference on Computer Vision and Pattern Recognizion (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"5_CR7","doi-asserted-by":"crossref","unstructured":"Chhipa, P.C., Holmgren, J.R., De, K., Saini, R., Liwicki, M.: Can self-supervised representation learning methods with stand distribution shifts and corruptions? In: 2023 IEEE\/CVF International Conference on Computer Vision Workshops. Workshop and Challenges for Out-of-Distribution Generalization in Computer Vision), pp. 4467\u20134476 (2023)","DOI":"10.1109\/ICCVW60793.2023.00481"},{"key":"5_CR8","unstructured":"Cho, H.C., Jhoo, W.Y., Kang, W., Roh, B.: Open-vocabulary object detection using pseudo caption labels. arXiv preprint arXiv:2303.13040 (2023)"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"De, K., Pedersen, M.: Impact of colour on robustness of deep neural networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 21\u201330 (2021)","DOI":"10.1109\/ICCVW54120.2021.00009"},{"key":"5_CR10","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"5_CR11","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Du, Y., Wei, F., Zhang, Z., Shi, M., Gao, Y., Li, G.: Learning to prompt for open-vocabulary object detection with vision-language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14084\u201314093 (2022)","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Feng, C., et al.: Promptdet: towards open-vocabulary detection using uncurated images. In: European Conference on Computer Vision, pp. 701\u2013717. Springer (2022)","DOI":"10.1007\/978-3-031-20077-9_41"},{"key":"5_CR14","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., et\u00a0al.: The many faces of robustness: a critical analysis of out-of-distribution generalization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8340\u20138349 (2021)","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"5_CR16","unstructured":"Hendrycks, D., Carlini, N., Schulman, J., Steinhardt, J.: Unsolved problems in ML safety. arXiv preprint arXiv:2109.13916 (2021)"},{"key":"5_CR17","unstructured":"Hendrycks, D., Dietterich, T.: Benchmarking neural network robustness to common corruptions and perturbations. In: Proceedings of the International Conference on Learning Representations (2019)"},{"key":"5_CR18","unstructured":"Hendrycks, D., Dietterich, T.G.: Benchmarking neural network robustness to common corruptions and surface variations. arXiv preprint arXiv:1807.01697 (2018)"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., Song, D.: Natural adversarial examples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15262\u201315271 (2021)","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"5_CR20","unstructured":"Idrissi, B.Y., et al.: Imagenet-x: understanding model mistakes with factor of variation annotations. arXiv preprint arXiv:2211.01866 (2022)"},{"key":"5_CR21","unstructured":"Kaul, P., Xie, W., Zisserman, A.: Multi-modal classifiers for open-vocabulary object detection. In: International Conference on Machine Learning, pp. 15946\u201315969. PMLR (2023)"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Kim, D., Angelova, A., Kuo, W.: Contrastive feature masking open-vocabulary vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15602\u201315612 (2023)","DOI":"10.1109\/ICCV51070.2023.01430"},{"key":"5_CR23","doi-asserted-by":"crossref","unstructured":"Kim, D., Angelova, A., Kuo, W.: Region-aware pretraining for open-vocabulary object detection with vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11144\u201311154 (2023)","DOI":"10.1109\/CVPR52729.2023.01072"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"5_CR25","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, vol. 25 (2012)"},{"key":"5_CR26","unstructured":"Kuo, W., Cui, Y., Gu, X., Piergiovanni, A., Angelova, A.: F-VLM: open-vocabulary object detection upon frozen vision and language models. arXiv preprint arXiv:2209.15639 (2022)"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Li*, L.H., et al.: Grounded language-image pre-training. In: CVPR (2022)","DOI":"10.1109\/CVPR52729.2023.02240"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Li, X., Chen, Y., Zhu, Y., Wang, S., Zhang, R., Xue, H.: Imagenet-e: benchmarking neural network robustness via attribute editing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20371\u201320381 (2023)","DOI":"10.1109\/CVPR52729.2023.01951"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"5_CR30","doi-asserted-by":"crossref","unstructured":"Liu, S., et\u00a0al.: Grounding DINO: marrying DINO with grounded pre-training for open-set object detection. In: European Conference on Computer Vision (ECCV) (2024)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Malik, H.S., Huzaifa, M., Naseer, M., Khan, S., Khan, F.S.: Objectcompose: evaluating resilience of vision-based models on object-to-background compositional changes. arXiv preprint arXiv:2403.04701 (2024)","DOI":"10.1007\/978-981-96-0917-8_23"},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Mao, X., et al.: COCO-O: a benchmark for object detectors under natural distribution shifts. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6339\u20136350 (2023)","DOI":"10.1109\/ICCV51070.2023.00583"},{"key":"5_CR34","unstructured":"Michaelis, C., et al.: Benchmarking robustness in object detection: autonomous driving when winter is coming. arXiv preprint arXiv:1907.07484 (2019)"},{"key":"5_CR35","unstructured":"Minderer, M., Gritsenko, A., Houlsby, N.: Scaling open-vocabulary object detection. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"5_CR36","doi-asserted-by":"crossref","unstructured":"Minderer, M., et\u00a0al.: Simple open-vocabulary object detection. In: European Conference on Computer Vision (ECCV), pp. 728\u2013755. Springer (2022)","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"5_CR37","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"5_CR38","unstructured":"Ramesh, A., et al.: Zero-shot text-to-image generation. In: International Conference on Machine Learning, pp. 8821\u20138831. PMLR (2021)"},{"key":"5_CR39","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 779\u2013788 (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"5_CR40","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"5_CR41","doi-asserted-by":"crossref","unstructured":"Shi, H., Hayat, M., Cai, J.: Open-vocabulary object detection via scene graph discovery. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 4012\u20134021 (2023)","DOI":"10.1145\/3581783.3612407"},{"key":"5_CR42","unstructured":"Wang, H., Ge, S., Lipton, Z., Xing, E.P.: Learning robust global representations by penalizing local predictive power. In: Advances in Neural Information Processing Systems, pp. 10506\u201310518 (2019)"},{"key":"5_CR43","doi-asserted-by":"crossref","unstructured":"Wang, L., et al.: Object-aware distillation pyramid for open-vocabulary object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11186\u201311196 (2023)","DOI":"10.1109\/CVPR52729.2023.01076"},{"key":"5_CR44","doi-asserted-by":"crossref","unstructured":"Yao, L., et al.: Detclipv2: scalable open-vocabulary object detection pre-training via word-region alignment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23497\u201323506 (2023)","DOI":"10.1109\/CVPR52729.2023.02250"},{"key":"5_CR45","doi-asserted-by":"crossref","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., Loy, C.C.: Open-vocabulary DETR with conditional matching. In: European Conference on Computer Vision, pp. 106\u2013122. Springer (2022)","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"5_CR46","doi-asserted-by":"crossref","unstructured":"Zareian, A., Rosa, K.D., Hu, D.H., Chang, S.F.: Open-vocabulary object detection using captions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14393\u201314402 (2021)","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"5_CR47","unstructured":"Zhao, S., et\u00a0al.: Improving pseudo labels for open-vocabulary object detection. arXiv preprint arXiv:2308.06412 (2023)"},{"key":"5_CR48","doi-asserted-by":"crossref","unstructured":"Zhao, S., et al.: Exploiting unlabeled data with vision and language models for object detection. In: European Conference on Computer Vision, pp. 159\u2013175. Springer (2022)","DOI":"10.1007\/978-3-031-20077-9_10"},{"key":"5_CR49","doi-asserted-by":"crossref","unstructured":"Zhong, Y., et\u00a0al.: Regionclip: region-based language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16793\u201316803 (2022)","DOI":"10.1109\/CVPR52688.2022.01629"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91672-4_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,20]],"date-time":"2025-05-20T15:25:40Z","timestamp":1747754740000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91672-4_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031916717","9783031916724"],"references-count":49,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91672-4_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}