{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:20:39Z","timestamp":1783527639976,"version":"3.55.0"},"reference-count":29,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T00:00:00Z","timestamp":1771459200000},"content-version":"vor","delay-in-days":49,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Procedia Computer Science"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1016\/j.procs.2026.02.133","type":"journal-article","created":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T07:17:59Z","timestamp":1774250279000},"page":"919-927","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Prompt Tuning and Retrieval in Open-Vocabulary Object Detection for Configurable Robot Vision"],"prefix":"10.1016","volume":"277","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0227-7017","authenticated-orcid":false,"given":"Stefan","family":"Fixl","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2321-2543","authenticated-orcid":false,"given":"Michael","family":"Hofmann","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4312-2369","authenticated-orcid":false,"given":"Andreas","family":"Pichler","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.procs.2026.02.133_bib1","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S., 2020. End-to-end object detection with transformers, in:Computer Vision\u2013ECCV 2020, Springer International Publishing, Cham. pp. 213\u2013229. doi: 10.1007\/978-3-030-58452-8_13.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"10.1016\/j.procs.2026.02.133_bib2","doi-asserted-by":"crossref","unstructured":"Cheng, D., Huang, S., Bi, J., Zhan, Y., Liu, J., Wang, Y., Sun, H., Wei, F., Deng, W., Zhang, Q., 2023. UPRISE: Universal prompt retrieval for improving zero-shot evaluation, in: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, Association for Computational Linguistics, Singapore. pp. 12318\u201312337. doi: 10.18653\/v1\/2023.emnlp-main.758.","DOI":"10.18653\/v1\/2023.emnlp-main.758"},{"key":"10.1016\/j.procs.2026.02.133_bib3","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K., 2019. BERT: Pre-training of deep bidirectional transformers for language understanding, in: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), Association for Computational Linguistics, Minneapolis, Minnesota. pp. 4171\u20134186. doi: 10.18653\/v1\/N19-1423.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.procs.2026.02.133_bib4","unstructured":"Douze, M., Guzhva, A., Deng, C., Johnson, J., Szilvasy, G., Mazar\u00e9, P.E., Lomeli, M., Hosseini, L., J\u00e9gou, H., 2024. The faiss library. arXiv preprint arXiv:2401.08281."},{"key":"10.1016\/j.procs.2026.02.133_bib5","series-title":"Lvis: A dataset for large vocabulary instance segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Gupta","year":"2019"},{"key":"10.1016\/j.procs.2026.02.133_bib6","doi-asserted-by":"crossref","unstructured":"Jiang, Q., Li, F., Zeng, Z., Ren, T., Liu, S., Zhang, L., 2025. T-rex2: Towards generic object detection via text-visual prompt synergy, in:Computer Vision\u2013ECCV 2024, Springer Nature Switzerland, Cham. pp. 38\u201357. doi: 10.1007\/978-3-031-73414-4_3.","DOI":"10.1007\/978-3-031-73414-4_3"},{"key":"10.1016\/j.procs.2026.02.133_bib7","doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N., 2021. Mdetr - modulated detection for end-to-end multi-modal understanding, in: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1760\u20131770. doi: 10.1109\/ICCV48922.2021.00180.","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"10.1016\/j.procs.2026.02.133_bib8","doi-asserted-by":"crossref","unstructured":"Kim, J., Cho, E., Kim, S., Kim, H.J., 2024. Retrieval-augmented open-vocabulary object detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17427\u201317436.","DOI":"10.1109\/CVPR52733.2024.01650"},{"key":"10.1016\/j.procs.2026.02.133_bib9","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., Constant, N., 2021. The power of scale for parameter-efficient prompt tuning, in: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, Association for Computational Linguistics. doi: 10.18653\/v1\/2021.emnlp-main.243.","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"10.1016\/j.procs.2026.02.133_bib10","doi-asserted-by":"crossref","unstructured":"Li, G., Zhang, M., Zheng, X., Chen, P., Wang, Z., Shen, Y., Zhuge, M., Wu, C., Chao, F., Li, K., Sun, X., Ji, R., 2024. Multi-modal inplace prompt tuning for open-set object detection, in: Proceedings of the 32nd ACM International Conference on Multimedia, Association for Computing Machinery, New York, NY, USA. p. 8062\u20138071. URL: https:\/\/doi.org\/10.1145\/3664647.3681275, doi: 10.1145\/3664647.3681275.","DOI":"10.1145\/3664647.3681275"},{"key":"10.1016\/j.procs.2026.02.133_bib11","doi-asserted-by":"crossref","unstructured":"Li, L.H., Zhang, P., Zhang, H., Yang, J., Li, C., Zhong, Y., Wang, L., Yuan, L., Zhang, L., Hwang, J.N., Chang, K.W., Gao, J., 2022. Grounded language-image pre-training, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10965\u201310975.","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"10.1016\/j.procs.2026.02.133_bib12","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., Zhu, J., Zhang, L., 2025. Grounding dino: Marrying dino with grounded pre-training for open-set object detection, in: Computer Vision\u2013ECCV 2024, Springer Nature Switzerland, Cham. pp. 38\u201355. doi: 10.1007\/978-3-031-72970-6_3.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"10.1016\/j.procs.2026.02.133_bib13","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B., 2021. Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.procs.2026.02.133_bib14","doi-asserted-by":"crossref","unstructured":"L\u00fcddecke, T., Ecker, A., 2022. Image segmentation using text and image prompts, in: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7076\u20137086. doi: 10.1109\/CVPR52688.2022.00695.","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"10.1016\/j.procs.2026.02.133_bib15","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A., Houlsby, N., 2023. Scaling open-vocabulary object detection, in: Advances in Neural Information Processing Systems, Curran Associates, Inc.. pp. 72983\u201373007.","DOI":"10.52202\/075280-3191"},{"key":"10.1016\/j.procs.2026.02.133_bib16","doi-asserted-by":"crossref","unstructured":"Minderer, M., Gritsenko, A., Stone, A., Neumann, M., Weissenborn, D., Dosovitskiy, A., Mahendran, A., Arnab, A., Dehghani, M., Shen, Z., Wang, X., Zhai, X., Kipf, T., Houlsby, N., 2022. Simple open-vocabulary object detection, in: Computer Vision\u2013ECCV 2022, Springer Nature Switzerland, Cham. pp. 728\u2013755. doi: 10.1007\/978-3-031-20080-9_42.","DOI":"10.1007\/978-3-031-20080-9_42"},{"key":"10.1016\/j.procs.2026.02.133_bib17","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., Sutskever, I., 2021. Learning transferable visual models from natural language supervision, in: Proceedings of the 38th International Conference on Machine Learning, PMLR. pp. 8748\u20138763."},{"key":"10.1016\/j.procs.2026.02.133_bib18","unstructured":"Ren, T., Jiang, Q., Liu, S., Zeng, Z., Liu, W., Gao, H., Huang, H., Ma, Z., Jiang, X., Chen, Y., Xiong, Y., Zhang, H., Li, F., Tang, P., Yu, K., Zhang, L., 2024. Grounding dino 1.5: Advance the \u201dedge\u201d of open-set object detection. arXiv preprint arXiv:2405.10300."},{"key":"10.1016\/j.procs.2026.02.133_bib19","unstructured":"Rong, J., Chen, H., Ou, L., Chen, T., Yu, X., Liu, Y., 2023. Retrieval-enhanced visual prompt learning for few-shot classification. arXiv preprint arXiv:2306.02243."},{"key":"10.1016\/j.procs.2026.02.133_bib20","unstructured":"Shazeer, N., Stern, M., 2018. Adafactor: Adaptive learning rates with sublinear memory cost, in: Proceedings of the 35th International Conference on Machine Learning, PMLR. pp. 4596\u20134604."},{"key":"10.1016\/j.procs.2026.02.133_bib21","unstructured":"Wang, H., Ren, P., Jie, Z., Dong, X., Feng, C., Qian, Y., Ma, L., Jiang, D., Wang, Y., Lan, X., et al., 2024. Ov-dino: Unified open-vocabulary detection with language-aware selective fusion. arXiv preprint arXiv:2407.07844."},{"key":"10.1016\/j.procs.2026.02.133_bib22","doi-asserted-by":"crossref","unstructured":"Wolf, T., Debut, L., Sanh, V., Chaumond, J., Delangue, C., Moi, A., Cistac, P., Rault, T., Louf, R., Funtowicz, M., Davison, J., Shleifer, S., von Platen, P., Ma, C., Jernite, Y., Plu, J., Xu, C., Scao, T.L., Gugger, S., Drame, M., Lhoest, Q., Rush, A.M., 2020. Transformers: State-of-the-art natural language processing, in: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, Association for Computational Linguistics. pp. 38\u201345. doi: 10.18653\/v1\/2020.emnlp-demos.6.","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"10.1016\/j.procs.2026.02.133_bib23","doi-asserted-by":"crossref","unstructured":"Xu, Y., Zhang, M., Fu, C., Chen, P., Yang, X., Li, K., Xu, C., 2023. Multi-modal queried object detection in the wild, in: Advances in Neural Information Processing Systems, Curran Associates, Inc.. pp. 4452\u20134469.","DOI":"10.52202\/075280-0198"},{"key":"10.1016\/j.procs.2026.02.133_bib24","doi-asserted-by":"crossref","unstructured":"Yao, L., Han, J., Liang, X., Xu, D., Zhang, W., Li, Z., Xu, H., 2023. Detclipv2: Scalable open-vocabulary object detection pre-training via word-region alignment, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23497\u201323506.","DOI":"10.1109\/CVPR52729.2023.02250"},{"key":"10.1016\/j.procs.2026.02.133_bib25","doi-asserted-by":"crossref","unstructured":"Yao, L., Han, J., Wen, Y., Liang, X., Xu, D., Zhang, W., Li, Z., XU, C., Xu, H., 2022. Detclip: Dictionary-enriched visual-concept paralleled pre-training for open-world detection, in: Advances in Neural Information Processing Systems, Curran Associates, Inc.. pp. 9125\u20139138.","DOI":"10.52202\/068431-0663"},{"key":"10.1016\/j.procs.2026.02.133_bib26","doi-asserted-by":"crossref","unstructured":"Yao, L., Pi, R., Han, J., Liang, X., Xu, H., Zhang, W., Li, Z., Xu, D., 2024. Detclipv3: Towards versatile generative open-vocabulary object detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 27391\u201327401.","DOI":"10.1109\/CVPR52733.2024.02586"},{"key":"10.1016\/j.procs.2026.02.133_bib27","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L.M., Shum, H.Y., 2022a. Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605."},{"key":"10.1016\/j.procs.2026.02.133_bib28","doi-asserted-by":"crossref","unstructured":"Zhang, H., Zhang, P., Hu, X., Chen, Y.C., Li, L., Dai, X., Wang, L., Yuan, L., Hwang, J.N., Gao, J., 2022b. Glipv2: Unifying localization and vision-language understanding, in: Advances in Neural Information Processing Systems, Curran Associates, Inc.. pp. 36067\u201336080.","DOI":"10.52202\/068431-2614"},{"key":"10.1016\/j.procs.2026.02.133_bib29","unstructured":"Zuwei Long, W.L., 2023. Open grounding dino: The third party implementation of the paper grounding dino. [Online] https:\/\/github.com\/longzw1997\/Open-GroundingDino."}],"container-title":["Procedia Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1877050926002498?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1877050926002498?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T15:42:07Z","timestamp":1783525327000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1877050926002498"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":29,"alternative-id":["S1877050926002498"],"URL":"https:\/\/doi.org\/10.1016\/j.procs.2026.02.133","relation":{},"ISSN":["1877-0509"],"issn-type":[{"value":"1877-0509","type":"print"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Prompt Tuning and Retrieval in Open-Vocabulary Object Detection for Configurable Robot Vision","name":"articletitle","label":"Article Title"},{"value":"Procedia Computer Science","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.procs.2026.02.133","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier B.V.","name":"copyright","label":"Copyright"}]}}