{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T12:13:55Z","timestamp":1775132035359,"version":"3.50.1"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Key Science and Technology Program of Henan Province","award":["No.252102210091 and 252102220120"],"award-info":[{"award-number":["No.252102210091 and 252102220120"]}]},{"name":"University Young Backbone Teachers Program of Henan Province","award":["2023GGJS053"],"award-info":[{"award-number":["2023GGJS053"]}]},{"name":"Fundamental Research Funds for the Universities of Henan Province","award":["NSFRF220414"],"award-info":[{"award-number":["NSFRF220414"]}]},{"name":"Excellent Young Teachers Program of Henan Polytechnic University","award":["No.2019XQG - 02"],"award-info":[{"award-number":["No.2019XQG - 02"]}]},{"name":"theNaturalScienceFoundationofHenanProvince","award":["No.242300420284)"],"award-info":[{"award-number":["No.242300420284)"]}]},{"name":"NationalNaturalScienceFoundationof China","award":["No.62472145"],"award-info":[{"award-number":["No.62472145"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s00530-026-02211-2","type":"journal-article","created":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T10:12:48Z","timestamp":1770113568000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards universal object detection: a fine-grained perspective on open-vocabulary object detection"],"prefix":"10.1007","volume":"32","author":[{"given":"Jing","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghua","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhanqiang","family":"Huo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingxu","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,3]]},"reference":[{"key":"2211_CR1","doi-asserted-by":"crossref","unstructured":"Zhao, X., Li, X., Duan, H., Huang, H., Li, Y., Chen, K., Yang, H.: Mg-llava: Towards multi-granularity visual instruction tuning. arXiv preprint arXiv:2406.17770 (2024)","DOI":"10.1109\/TCSVT.2025.3643469"},{"key":"2211_CR2","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., : Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In: European Conference on Computer Vision, pp. 38\u201355 (2025). Springer","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"2211_CR3","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., : Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2211_CR4","unstructured":"Liu, Z., Wang, Y., Vaidya, S., Ruehle, F., Halverson, J., Solja\u010di\u0107, M., Hou, T.Y., Tegmark, M.: Kan: Kolmogorov-arnold networks. arXiv preprint arXiv:2404.19756 (2024)"},{"key":"2211_CR5","unstructured":"Wang, H., Ren, P., Jie, Z., Dong, X., Feng, C., Qian, Y., Ma, L., Jiang, D., Wang, Y., Lan, X., Liang, X.: Ov-dino: Unified open-vocabulary detection with language-aware selective fusion. arXiv preprint arXiv:2407.07844 (2024)"},{"key":"2211_CR6","unstructured":"Bharadwaj, R., Naseer, M., Khan, S., Khan, F.S.: Enhancing novel object detection via cooperative foundational models. arXiv preprint arXiv:2311.12068 (2023)"},{"key":"2211_CR7","doi-asserted-by":"crossref","unstructured":"Wu, Z., Gao, J., Xu, C.: Open-vocabulary video scene graph generation via union-aware semantic alignment. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 8566\u20138575 (2024)","DOI":"10.1145\/3664647.3681061"},{"key":"2211_CR8","doi-asserted-by":"crossref","unstructured":"Joseph, K., Khan, S., Khan, F.S., Balasubramanian, V.N.: Towards open world object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5830\u20135840 (2021)","DOI":"10.1109\/CVPR46437.2021.00577"},{"key":"2211_CR9","doi-asserted-by":"crossref","unstructured":"Zhao, X., Ma, Y., Wang, D., Shen, Y., Qiao, Y., Liu, X.: Revisiting open world object detection. IEEE Trans. Circuits and Syst. Video Technol. (2023)","DOI":"10.1109\/TCSVT.2023.3326279"},{"key":"2211_CR10","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110472","volume":"152","author":"Y Chen","year":"2024","unstructured":"Chen, Y., Ma, L., Jing, L., Yu, J.: Bsdp: Brain-inspired streaming dual-level perturbations for online open world object detection. Pattern Recogn. 152, 110472 (2024)","journal-title":"Pattern Recogn."},{"key":"2211_CR11","doi-asserted-by":"crossref","unstructured":"Wu, Z., Lu, Y., Chen, X., Wu, Z., Kang, L., Yu, J.: Uc-owod: Unknown-classified open world object detection. In: European Conference on Computer Vision, pp. 193\u2013210 (2022). Springer","DOI":"10.1007\/978-3-031-20080-9_12"},{"key":"2211_CR12","unstructured":"Pershouse, D., Dayoub, F., Miller, D., S\u00fcnderhauf, N.: Addressing the challenges of open-world object detection. arXiv preprint arXiv:2303.14930 (2023)"},{"key":"2211_CR13","doi-asserted-by":"publisher","first-page":"82560","DOI":"10.52202\/079017-2625","volume":"37","author":"M Chen","year":"2024","unstructured":"Chen, M., Gao, J., Xu, C.: Conjugated semantic pool improves ood detection with pre-trained vision-language models. Adv. Neural. Inf. Process. Syst. 37, 82560\u201382593 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2211_CR14","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A., Stone, A., Neumann, M., Weissenborn, D., Dosovitskiy, A., Mahendran, A., Arnab, A., Dehghani, M., Shen, Z., : Simple open-vocabulary object detection. In: European Conference on Computer Vision, pp. 728\u2013755 (2022). Springer","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"2211_CR15","doi-asserted-by":"crossref","unstructured":"Maaz, M., Rasheed, H., Khan, S., Khan, F.S., Anwer, R.M., Yang, M.-H.: Class-agnostic object detection with multi-modal transformer. In: European Conference on Computer Vision, pp. 512\u2013531 (2022). Springer","DOI":"10.1007\/978-3-031-20080-9_30"},{"key":"2211_CR16","doi-asserted-by":"crossref","unstructured":"Bravo, M.A., Mittal, S., Brox, T.: Localized vision-language matching for open-vocabulary object detection. In: DAGM German Conference on Pattern Recognition, pp. 393\u2013408 (2022). Springer","DOI":"10.1007\/978-3-031-16788-1_24"},{"key":"2211_CR17","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A., Houlsby, N.: Scaling open-vocabulary object detection. Adv. Neural Inform. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-3191"},{"key":"2211_CR18","doi-asserted-by":"crossref","unstructured":"Cho, J.H., Kr\u00e4henb\u00fchl, P.: Language-conditioned detection transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16593\u201316603 (2024)","DOI":"10.1109\/CVPR52733.2024.01570"},{"key":"2211_CR19","doi-asserted-by":"crossref","unstructured":"Gao, J., Chen, M., Xu, C.: Learning probabilistic presence-absence evidence for weakly-supervised audio-visual event perception. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025)","DOI":"10.1109\/TPAMI.2025.3546312"},{"key":"2211_CR20","doi-asserted-by":"crossref","unstructured":"Kim, D., Angelova, A., Kuo, W.: Region-aware pretraining for open-vocabulary object detection with vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11144\u201311154 (2023)","DOI":"10.1109\/CVPR52729.2023.01072"},{"key":"2211_CR21","doi-asserted-by":"crossref","unstructured":"Zareian, A., Rosa, K.D., Hu, D.H., Chang, S.-F.: Open-vocabulary object detection using captions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14393\u201314402 (2021)","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"2211_CR22","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755 (2014). Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2211_CR23","doi-asserted-by":"crossref","unstructured":"Gupta, A., Dollar, P., Girshick, R.: Lvis: A dataset for large vocabulary instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5356\u20135364 (2019)","DOI":"10.1109\/CVPR.2019.00550"},{"key":"2211_CR24","unstructured":"Gu, X., Lin, T.-Y., Kuo, W., Cui, Y.: Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:2104.13921 (2021)"},{"key":"2211_CR25","doi-asserted-by":"crossref","unstructured":"Zhou, X., Girdhar, R., Joulin, A., Kr\u00e4henb\u00fchl, P., Misra, I.: Detecting twenty-thousand classes using image-level supervision. In: ECCV (2022)","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"2211_CR26","doi-asserted-by":"crossref","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., Loy, C.C.: Open-vocabulary detr with conditional matching. In: European Conference on Computer Vision, pp. 106\u2013122 (2022). Springer","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"2211_CR27","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Yang, J., Zhang, P., Li, C., Codella, N., Li, L.H., Zhou, L., Dai, X., Yuan, L., Li, Y., : Regionclip: Region-based language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16793\u201316803 (2022)","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"2211_CR28","unstructured":"Lin, C., Sun, P., Jiang, Y., Luo, P., Qu, L., Haffari, G., Yuan, Z., Cai, J.: Learning object-language alignments for open-vocabulary object detection. arXiv preprint arXiv:2211.14843 (2022)"},{"key":"2211_CR29","unstructured":"Chen, P., Sheng, K., Zhang, M., Lin, M., Shen, Y., Lin, S., Ren, B., Li, K.: Open vocabulary object detection with proposal mining and prediction equalization. arXiv preprint arXiv:2206.11134 (2022)"},{"key":"2211_CR30","doi-asserted-by":"crossref","unstructured":"Wu, S., Zhang, W., Jin, S., Liu, W., Loy, C.C.: Aligning bag of regions for open-vocabulary object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15254\u201315264 (2023)","DOI":"10.1109\/CVPR52729.2023.01464"},{"key":"2211_CR31","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhang, Z., Schulter, S., Zhao, L., Vijay\u00a0Kumar, B., Stathopoulos, A., Chandraker, M., Metaxas, D.N.: Exploiting unlabeled data with vision and language models for object detection. In: European Conference on Computer Vision, pp. 159\u2013175 (2022). Springer","DOI":"10.1007\/978-3-031-20077-9_10"},{"key":"2211_CR32","doi-asserted-by":"crossref","unstructured":"Wu, S., Zhang, W., Xu, L., Jin, S., Liu, W., Loy, C.C.: Clim: Contrastive language-image mosaic for region representation. In: Proceedings of the AAAI Conference on Artificial Intelligence 38:6117\u20136125 (2024)","DOI":"10.1609\/aaai.v38i6.28428"},{"key":"2211_CR33","doi-asserted-by":"crossref","unstructured":"Zhao, S., Schulter, S., Zhao, L., Zhang, Z., Suh, Y., Chandraker, M., Metaxas, D.N., : Taming self-training for open-vocabulary object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13938\u201313947 (2024)","DOI":"10.1109\/CVPR52733.2024.01322"},{"key":"2211_CR34","unstructured":"Song, H., Bang, J.: Prompt-guided transformers for end-to-end open-vocabulary object detection. arXiv preprint arXiv:2303.14386 (2023)"},{"key":"2211_CR35","doi-asserted-by":"crossref","unstructured":"Kim, D., Angelova, A., Kuo, W.: Contrastive feature masking open-vocabulary vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15602\u201315612 (2023)","DOI":"10.1109\/ICCV51070.2023.01430"},{"key":"2211_CR36","unstructured":"Wu, S., Zhang, W., Xu, L., Jin, S., Li, X., Liu, W., Loy, C.C.: Clipself: Vision transformer distills itself for open-vocabulary dense prediction. arXiv preprint arXiv:2310.01403 (2023)"},{"key":"2211_CR37","doi-asserted-by":"crossref","unstructured":"Wang, J., Chen, B., Kang, B., Li, Y., Chen, Y., Xian, W., Chang, H., Xu, Y.: Ov-dquo: Open-vocabulary detr with denoising text query training and open-world unknown objects supervision. arXiv preprint arXiv:2405.17913 (2024)","DOI":"10.1609\/aaai.v39i7.32836"},{"key":"2211_CR38","unstructured":"Kuo, W., Cui, Y., Gu, X., Piergiovanni, A., Angelova, A.: F-vlm: Open-vocabulary object detection upon frozen vision and language models. arXiv preprint arXiv:2209.15639 (2022)"},{"key":"2211_CR39","doi-asserted-by":"crossref","unstructured":"Wu, X., Zhu, F., Zhao, R., Li, H.: Cora: Adapting clip for open-vocabulary detection with region prompting and anchor pre-matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7031\u20137040 (2023)","DOI":"10.1109\/CVPR52729.2023.00679"},{"key":"2211_CR40","doi-asserted-by":"crossref","unstructured":"Ma, C., Jiang, Y., Wen, X., Yuan, Z., Qi, X.: Codet: Co-occurrence guided region-word alignment for open-vocabulary object detection. Adv. Neural Inform. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-3113"},{"key":"2211_CR41","doi-asserted-by":"crossref","unstructured":"Hou, X., Liu, M., Zhang, S., Wei, P., Chen, B.: Salience detr: Enhancing detection transformer with hierarchical salience filtering refinement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17574\u201317583 (2024)","DOI":"10.1109\/CVPR52733.2024.01664"},{"key":"2211_CR42","doi-asserted-by":"crossref","unstructured":"Hou, X., Liu, M., Zhang, S., Wei, P., Chen, B., Lan, X.: Relation detr: Exploring explicit position relation prior for object detection. In: European Conference on Computer Vision, pp. 89\u2013105 (2025). Springer","DOI":"10.1007\/978-3-031-72973-7_6"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02211-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02211-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02211-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T11:36:37Z","timestamp":1775129797000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02211-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,3]]},"references-count":42,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["2211"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02211-2","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,3]]},"assertion":[{"value":"26 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"All authors agreed to participate in this paper.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This article is licensed under a Creative Commons Attribution 4.0 International License, which permits use, sharing, adaptation, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if changes were made. The images or other third party material in this article are included in the article\u2019s Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article\u2019s Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit\n                      \n                      .","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Open Access"}}],"article-number":"139"}}