{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T08:55:13Z","timestamp":1770800113302,"version":"3.50.0"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T00:00:00Z","timestamp":1764979200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T00:00:00Z","timestamp":1764979200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62376197"],"award-info":[{"award-number":["62376197"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62376197"],"award-info":[{"award-number":["62376197"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62376197"],"award-info":[{"award-number":["62376197"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s00530-025-02112-w","type":"journal-article","created":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T09:17:06Z","timestamp":1765012626000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Open-vocabulary object detection via prompt learning and dual-branch classification"],"prefix":"10.1007","volume":"32","author":[{"given":"Yao","family":"Xiao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dexin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"2112_CR1","doi-asserted-by":"crossref","unstructured":"Zareian, A., Rosa, K.D., Hu, D.H., Chang, S.F.: Open-vocabulary object detection using captions, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14393\u201314402 (2021)","DOI":"10.1109\/CVPR46437.2021.01416"},{"issue":"8","key":"2112_CR2","doi-asserted-by":"publisher","first-page":"5625","DOI":"10.1109\/TPAMI.2024.3369699","volume":"46","author":"J Zhang","year":"2024","unstructured":"Zhang, J., Huang, J., Jin, S., Lu, S.: Vision-language models for vision tasks: A survey. IEEE Trans. Pattern Anal. Mach. Intell. 46(8), 5625\u201344 (2024)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2112_CR3","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision, in: International conference on machine learning, PMLR, pp. 8748\u20138763 (2021)"},{"key":"2112_CR4","unstructured":"Jia, C., Yang, Y., Xia, Y., Chen, Y.T., Parekh, Z., Pham, H., Le, Q., Sung, Y.H., Li, Z., Duerig, T.: Scaling up visual and vision-language representation learning with noisy text supervision, in: International conference on machine learning, PMLR, pp. 4904\u20134916 (2021)"},{"key":"2112_CR5","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130, 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"2112_CR6","unstructured":"Gu, X., Lin, T.Y., Kuo, W., Cui, Y.: Open-vocabulary object detection via vision and language knowledge distillation, International Conference on Learning Representations (2021)"},{"key":"2112_CR7","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Yang, J., Zhang, P., Li, C., Codella, N., Li, L.H., Zhou, L., Dai, X., Yuan, L., Li, Y. et\u00a0al.: Regionclip: Region-based language-image pretraining, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16793\u201316803 (2022)","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"2112_CR8","doi-asserted-by":"crossref","unstructured":"Feng, C., Zhong, Y., Jie, Z., Chu, X., Ren, H., Wei, X., Xie, W., Ma, L.: Promptdet: Towards open-vocabulary detection using uncurated images, in: European conference on computer vision, Springer, pp. 701\u2013717 (2022)","DOI":"10.1007\/978-3-031-20077-9_41"},{"key":"2112_CR9","doi-asserted-by":"crossref","unstructured":"Bravo, M.A., Mittal, S., Brox, T. Localized vision-language matching for open-vocabulary object detection, in: DAGM German conference on pattern recognition, Springer, pp. 393\u2013408 (2022)","DOI":"10.1007\/978-3-031-16788-1_24"},{"key":"2112_CR10","doi-asserted-by":"crossref","unstructured":"Kim, D., Angelova, A., Kuo, W.: Region-aware pretraining for open-vocabulary object detection with vision transformers, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 11144\u201311154 (2023)","DOI":"10.1109\/CVPR52729.2023.01072"},{"key":"2112_CR11","unstructured":"Lin, C., Sun, P., Jiang, Y., Luo, P., Qu, L, Haffari, G., Yuan, Z, Cai, J.: Learning object-language alignments for open-vocabulary object detection, The Eleventh International Conference on Learning Representations (2022)"},{"key":"2112_CR12","doi-asserted-by":"crossref","unstructured":"Zhou, X., Girdhar, R., Joulin, A., Kr\u00e4henb\u00fchl P., Misra I.: Detecting twenty-thousand classes using image-level supervision, in: European Conference on Computer Vision, Springer, pp. 350\u2013368 (2022)","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"2112_CR13","doi-asserted-by":"crossref","unstructured":"Changpinyo, S., Sharma, P., Ding, N., Soricut, R.: Conceptual 12m: Pushing web-scale image-text pre-training to recognize long-tail visual concepts, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 3558\u20133568 (2021)","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"2112_CR14","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhang, Z., Schulter, S., Zhao, L., Vijay\u00a0Kumar, B., Stathopoulos A., Chandraker M., Metaxas, D.N. Exploiting unlabeled data with vision and language models for object detection, in: European conference on computer vision, Springer, pp. 159\u2013175 (2022)","DOI":"10.1007\/978-3-031-20077-9_10"},{"key":"2112_CR15","doi-asserted-by":"crossref","unstructured":"Gao, M., Xing, C., Niebles, J.C., Li, J., Xu, R., Liu, W., Xiong, C.: Open vocabulary object detection with pseudo bounding-box labels, in: European Conference on Computer Vision, Springer, pp. 266\u2013282 (2022)","DOI":"10.1007\/978-3-031-20080-9_16"},{"key":"2112_CR16","doi-asserted-by":"crossref","unstructured":"Wang, L., Liu, Y., Du, P., Ding, Z., Liao, Y., Qi, Q., Chen, B. and Liu, S.: Object-aware distillation pyramid for open-vocabulary object detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11186\u201311196 (2023)","DOI":"10.1109\/CVPR52729.2023.01076"},{"key":"2112_CR17","doi-asserted-by":"crossref","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., Loy C.C.: Open-vocabulary detr with conditional matching, in: European Conference on Computer Vision, Springer, pp. 106\u2013122 (2022)","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"2112_CR18","doi-asserted-by":"crossref","unstructured":"Wu, S., Zhang, W., Jin, S., Liu, W., Loy, C.C.: Aligning bag of regions for open-vocabulary object detection, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 15254\u201315264 (2023)","DOI":"10.1109\/CVPR52729.2023.01464"},{"key":"2112_CR19","unstructured":"Kuo, W., Cui, Y., Gu, X., Piergiovanni, A., Angelova, A.: F-vlm: Open-vocabulary object detection upon frozen vision and language models, The Eleventh International Conference on Learning Representations (2022)"},{"key":"2112_CR20","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A, Stone, A., Neumann, M., Weissenborn, D., Dosovitskiy, A., Mahendran, A., Arnab, A., Dehghani, M., Shen, Z. et\u00a0al.: Simple open-vocabulary object detection, in: European Conference on Computer Vision, Springer, pp. 728\u2013755 (2022)","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"2112_CR21","doi-asserted-by":"crossref","unstructured":"Wu, X., Zhu, F., Zhao, R., Li, H.: Cora: Adapting clip for open-vocabulary detection with region prompting and anchor pre-matching, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7031\u20137040 (2023)","DOI":"10.1109\/CVPR52729.2023.00679"},{"key":"2112_CR22","doi-asserted-by":"crossref","unstructured":"Du, Y., Wei, F., Zhang, Z., Shi, M., Gao, Y., Li, G.: Learning to prompt for open-vocabulary object detection with vision-language model, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 14084\u201314093 (2022)","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"2112_CR23","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., Shan, Y.: Yolo-world: Real-time open-vocabulary object detection, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16901\u201316911 (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"2112_CR24","unstructured":"Song, H., Bang, J.: Prompt-guided transformers for end-to-end open-vocabulary object detection, arXiv preprint arXiv:2303.14386 (2023)"},{"key":"2112_CR25","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, J., Li, J., Li, G., Liu, S., Lin, L., Li, G.: Learning background prompts to discover implicit knowledge for open vocabulary object detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16678\u201316687 (2024)","DOI":"10.1109\/CVPR52733.2024.01578"},{"key":"2112_CR26","unstructured":"Wang, Z., Li, A., Zhou, F., Li, Z., Dou, Q.: Open-vocabulary object detection with meta prompt representation and instance contrastive optimization, arXiv preprint arXiv:2403.09433 (2024)"},{"key":"2112_CR27","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110648","volume":"155","author":"H Song","year":"2024","unstructured":"Song, H., Bang, J.: Prompt-guided detr with roi-pruned masked attention for open-vocabulary object detection. Pattern Recogn. 155, 110648 (2024)","journal-title":"Pattern Recogn."},{"key":"2112_CR28","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r , P., Zitnick, C.L.: Microsoft coco: Common objects in context, in: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, Proceedings, Part V 13, Springer, 2014, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2112_CR29","doi-asserted-by":"crossref","unstructured":"Gupta, A., Dollar, P., Girshick, R.: Lvis: A dataset for large vocabulary instance segmentation, in: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 5356\u20135364 (2019)","DOI":"10.1109\/CVPR.2019.00550"},{"key":"2112_CR30","doi-asserted-by":"crossref","unstructured":"Shao, S., Li, Z., Zhang, T., Peng, C., Yu, G., Zhang, X., Li, J., Sun, J.: Objects365: A large-scale, high-quality dataset for object detection, in: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 8430\u20138439 (2019)","DOI":"10.1109\/ICCV.2019.00852"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02112-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-02112-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02112-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T04:19:39Z","timestamp":1770783579000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-02112-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":30,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2112"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-02112-w","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"8 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"39"}}