{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T18:58:50Z","timestamp":1785265130349,"version":"3.55.0"},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476094"],"award-info":[{"award-number":["62476094"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB4704900"],"award-info":[{"award-number":["2023YFB4704900"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhuhai Industry-University-Institute Cooperation Project","award":["2220004002460"],"award-info":[{"award-number":["2220004002460"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11760-026-05564-3","type":"journal-article","created":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T11:10:54Z","timestamp":1784027454000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A prompt-robust vision-language detector for accurate UAV inventory inspection in power-material warehouses"],"prefix":"10.1007","volume":"20","author":[{"given":"Yuefeng","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiliang","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lianfang","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"issue":"2","key":"5564_CR1","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1007\/s11263-019-01247-4","volume":"128","author":"L Liu","year":"2020","unstructured":"Liu, L., Ouyang, W., Wang, X., Fieguth, P.: Deep learning for generic object detection: A survey. Int. J. Comput. Vision 128(2), 261\u2013318 (2020). https:\/\/doi.org\/10.1007\/s11263-019-01247-4","journal-title":"Int. J. Comput. Vision"},{"issue":"1","key":"5564_CR2","doi-asserted-by":"publisher","first-page":"149","DOI":"10.3390\/rs16010149","volume":"16","author":"G Tang","year":"2024","unstructured":"Tang, G., Ni, J., Zhao, Y.: A survey of object detection for UAVs based on deep learning. Remote Sensing 16(1), 149 (2024). https:\/\/doi.org\/10.3390\/rs16010149","journal-title":"Remote Sensing"},{"key":"5564_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2025.103623","volume":"126","author":"S Danish","year":"2026","unstructured":"Danish, S., Sadeghi-Niaraki, A., Khan, S.U., Dang, L.M., Tightiz, L., Moon, H.: A comprehensive survey of vision-language models: Pretrained models, fine-tuning, prompt engineering, adapters, and benchmark datasets. Information Fusion 126, 103623 (2026). https:\/\/doi.org\/10.1016\/j.inffus.2025.103623","journal-title":"Information Fusion"},{"key":"5564_CR4","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., Sutskever, I.: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 139, pp. 8748\u20138763. (2021). (PMLR, Virtual Event)"},{"key":"5564_CR5","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: Proceedings of the 39th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 162, pp. 12888\u201312900. PMLR, Baltimore, MD, USA (2022)"},{"issue":"7","key":"5564_CR6","doi-asserted-by":"publisher","first-page":"5629","DOI":"10.1109\/TPAMI.2024.3361862","volume":"46","author":"J Wu","year":"2024","unstructured":"Wu, J., Li, X., Xu, S., Yuan, J., Ding, H., Yang, Y., Zhang, J.: Towards open vocabulary learning: A survey. IEEE Trans. Pattern Anal. Mach. Intell. 46(7), 5629\u20135653 (2024). https:\/\/doi.org\/10.1109\/TPAMI.2024.3361862","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"8","key":"5564_CR7","doi-asserted-by":"publisher","first-page":"557","DOI":"10.3390\/drones9080557","volume":"9","author":"Y Zhou","year":"2025","unstructured":"Zhou, Y., Li, J., Ou, C., Yan, D., Zhang, H., Xue, X.: Open-vocabulary object detection in UAV imagery: A review and future perspectives. Drones 9(8), 557 (2025). https:\/\/doi.org\/10.3390\/drones9080557","journal-title":"Drones"},{"issue":"8","key":"5564_CR8","doi-asserted-by":"publisher","first-page":"526","DOI":"10.3390\/drones7080526","volume":"7","author":"Z Zhang","year":"2023","unstructured":"Zhang, Z.: Drone-YOLO: An efficient neural network method for target detection in drone images. Drones 7(8), 526 (2023). https:\/\/doi.org\/10.3390\/drones7080526","journal-title":"Drones"},{"key":"5564_CR9","doi-asserted-by":"publisher","first-page":"1332","DOI":"10.1007\/s11227-025-07836-0","volume":"81","author":"J Zhang","year":"2025","unstructured":"Zhang, J., Gao, M., Song, L., Zhao, H., Li, W., Zhang, Z., Rong, C.: REA-YOLO for small object detection in UAV aerial images. J. Supercomput. 81, 1332 (2025). https:\/\/doi.org\/10.1007\/s11227-025-07836-0","journal-title":"J. Supercomput."},{"issue":"10","key":"5564_CR10","doi-asserted-by":"publisher","first-page":"1768","DOI":"10.3390\/rs17101768","volume":"17","author":"Z Wan","year":"2025","unstructured":"Wan, Z., Lan, Y., Xu, Z., Shang, K., Zhang, F.: DAU-YOLO: A lightweight and effective method for small object detection in UAV images. Remote Sensing 17(10), 1768 (2025). https:\/\/doi.org\/10.3390\/rs17101768","journal-title":"Remote Sensing"},{"issue":"9","key":"5564_CR11","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.3390\/rs17091556","volume":"17","author":"Z Li","year":"2025","unstructured":"Li, Z., Lian, S., Pan, D., Wang, Y., Liu, W.: AD-Det: Boosting object detection in UAV images with focused small objects and balanced tail classes. Remote Sensing 17(9), 1556 (2025). https:\/\/doi.org\/10.3390\/rs17091556","journal-title":"Remote Sensing"},{"key":"5564_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.129866","volume":"634","author":"Y Si","year":"2025","unstructured":"Si, Y., Xu, H., Zhu, X., Zhang, W., Dong, Y., Chen, Y., Li, H.: SCSA: Exploring the synergistic effects between spatial and channel attention. Neurocomputing 634, 129866 (2025). https:\/\/doi.org\/10.1016\/j.neucom.2025.129866","journal-title":"Neurocomputing"},{"key":"5564_CR13","doi-asserted-by":"crossref","unstructured":"He, S., Jiang, C., Dong, D., Ding, L.: SD-Conv: Towards the parameter-efficiency of dynamic convolution. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 6454\u20136463. (2023)","DOI":"10.1109\/WACV56688.2023.00639"},{"key":"5564_CR14","doi-asserted-by":"crossref","unstructured":"Chen, L., Gu, L., Zheng, D., Fu, Y.: Frequency-adaptive dilated convolution for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3414\u20133425. (2024)","DOI":"10.1109\/CVPR52733.2024.00328"},{"key":"5564_CR15","doi-asserted-by":"publisher","unstructured":"Xiong, Y., Li, Z., Chen, Y., Wang, F., Zhu, X., Luo, J., Wang, W., Lu, T., Li, H., Qiao, Y., Lu, L., Zhou, J., Dai, J.: Efficient deformable ConvNets: Rethinking dynamic and sparse operator for vision applications. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5652\u20135661 (2024). https:\/\/doi.org\/10.1109\/CVPR52733.2024.00540","DOI":"10.1109\/CVPR52733.2024.00540"},{"key":"5564_CR16","doi-asserted-by":"publisher","unstructured":"Chen, L., Gu, L., Li, L., Yan, C., Fu, Y.: Frequency dynamic convolution for dense image prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 30178\u201330188 (2025). https:\/\/doi.org\/10.1109\/CVPR52734.2025.02809","DOI":"10.1109\/CVPR52734.2025.02809"},{"key":"5564_CR17","doi-asserted-by":"publisher","unstructured":"Li, L.H., Zhang, P., Zhang, H., Yang, J., Li, C., Zhong, Y., Wang, L., Yuan, L., Zhang, L., Hwang, J.-N., Chang, K.-W., Gao, J.: Grounded Language-Image Pre-Training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10965\u201310975 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01069","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"5564_CR18","doi-asserted-by":"publisher","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., Zhu, J., Zhang, L.: Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection. In: Computer Vision \u2013 ECCV 2024. Lecture Notes in Computer Science, vol. 15105, pp. 38\u201355. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-72970-6_3","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"5564_CR19","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., Shan, Y.: YOLO-World: Real-time open-vocabulary object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16901\u201316911. (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"5564_CR20","doi-asserted-by":"crossref","unstructured":"Wang, A., Liu, L., Chen, H., Lin, Z., Han, J., Ding, G.: YOLOE: Real-time seeing anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 24591\u201324602. (2025)","DOI":"10.1109\/ICCV51701.2025.02280"},{"key":"5564_CR21","doi-asserted-by":"publisher","unstructured":"Li, Y., Guo, W., Yang, X., Liao, N., He, D., Zhou, J., Yu, W.: Toward open vocabulary aerial object detection with CLIP-activated student-teacher learning. In: Computer Vision \u2013 ECCV 2024. Lecture Notes in Computer Science, vol. 15144, pp. 431\u2013448. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-73016-0_25","DOI":"10.1007\/978-3-031-73016-0_25"},{"key":"5564_CR22","doi-asserted-by":"crossref","unstructured":"Huang, Z., Feng, Y., Liu, Z., Yang, S., Liu, Q., Wang, Y.: OpenRSD: Towards open-prompts for object detection in remote sensing images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 8384\u20138394 (2025)","DOI":"10.1109\/ICCV51701.2025.00785"},{"key":"5564_CR23","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1016\/j.neucom.2022.07.042","volume":"506","author":"Y-F Zhang","year":"2022","unstructured":"Zhang, Y.-F., Ren, W., Zhang, Z., Jia, Z., Wang, L., Tan, T.: Focal and efficient IoU loss for accurate bounding box regression. Neurocomputing 506, 146\u2013157 (2022). https:\/\/doi.org\/10.1016\/j.neucom.2022.07.042","journal-title":"Neurocomputing"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05564-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05564-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05564-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T18:16:57Z","timestamp":1785262617000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05564-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":23,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["5564"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05564-3","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"20 March 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 June 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no competing interests.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Code is available from the corresponding author upon reasonable request.","order":2,"name":"Ethics","label":"Code availability","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"488"}}