{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:25:12Z","timestamp":1786980312808,"version":"3.56.0"},"reference-count":63,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T00:00:00Z","timestamp":1764806400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T00:00:00Z","timestamp":1764806400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]},{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]},{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]},{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]},{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]},{"name":"Major Program (JD) of Hubei Province, China","award":["Grant 2023BAA017"],"award-info":[{"award-number":["Grant 2023BAA017"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s00530-025-02060-5","type":"journal-article","created":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T07:04:24Z","timestamp":1764831864000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["OF-DETR: an efficient end-to-end detector for tiny traffic targets in aerial images"],"prefix":"10.1007","volume":"32","author":[{"given":"Jie","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hanzhang","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feiyu","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxuan","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuaidi","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinghao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qixiang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Minchao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,4]]},"reference":[{"issue":"15","key":"2060_CR1","doi-asserted-by":"publisher","first-page":"4529","DOI":"10.1080\/01431161.2023.2197129","volume":"44","author":"J Feng","year":"2023","unstructured":"Feng, J., Wang, J., Qin, R.: Lightweight detection network for arbitrary-oriented vehicles in UAV imagery via precise positional information encoding and bidirectional feature fusion. Int. J. Remote Sens. 44(15), 4529\u20134558 (2023)","journal-title":"Int. J. Remote Sens."},{"issue":"2","key":"2060_CR2","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1007\/s11831-019-09321-3","volume":"27","author":"KV Sakhare","year":"2020","unstructured":"Sakhare, K.V., Tewari, T., Vyas, V.: Review of vehicle detection systems in advanced driver assistant systems. Arch. Comput. Methods Eng. 27(2), 591\u2013610 (2020)","journal-title":"Arch. Comput. Methods Eng."},{"issue":"14","key":"2060_CR3","doi-asserted-by":"publisher","first-page":"3240","DOI":"10.3390\/rs14143240","volume":"14","author":"X Luo","year":"2022","unstructured":"Luo, X., Wu, Y., Zhao, L.: YOLOD: a target detection method for UAV aerial imagery. Remote Sens. 14(14), 3240 (2022)","journal-title":"Remote Sens."},{"issue":"20","key":"2060_CR4","doi-asserted-by":"publisher","first-page":"4027","DOI":"10.3390\/rs13204027","volume":"13","author":"S Byun","year":"2021","unstructured":"Byun, S., Shin, I.-K., Moon, J., et al.: Road traffic monitoring from UAV images using deep learning networks. Remote Sens. 13(20), 4027 (2021). https:\/\/doi.org\/10.3390\/rs13204027","journal-title":"Remote Sens."},{"issue":"1","key":"2060_CR5","doi-asserted-by":"publisher","first-page":"5565589","DOI":"10.1155\/2021\/5565589","volume":"2021","author":"X Liu","year":"2021","unstructured":"Liu, X., Zhang, Z.: A vision-based target detection, tracking, and positioning algorithm for unmanned aerial vehicle. Wirel. Commun. Mob. Com. 2021(1), 5565589 (2021). https:\/\/doi.org\/10.1155\/2021\/5565589","journal-title":"Wirel. Commun. Mob. Com."},{"key":"2060_CR6","doi-asserted-by":"publisher","first-page":"1462","DOI":"10.1109\/TMM.2023.3234822","volume":"25","author":"Z Liu","year":"2023","unstructured":"Liu, Z., Shang, Y., Li, T., et al.: Robust multi-drone multi-target tracking to resolve target occlusion: a benchmark. IEEE Trans. Multimedia. 25, 1462\u20131476 (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3234822","journal-title":"IEEE Trans. Multimedia"},{"key":"2060_CR7","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Lv, W., Xu, S., et al.: Detrs beat yolos on real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16965\u201316974 (2024)","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"2060_CR8","doi-asserted-by":"crossref","unstructured":"Zhu, X., Lyu, S., Wang, X.: TPH-YOLOv5: improved YOLOv5 based on transformer prediction head for object detection on drone-captured scenarios. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2778\u20132788 (2021)","DOI":"10.1109\/ICCVW54120.2021.00312"},{"key":"2060_CR9","doi-asserted-by":"publisher","unstructured":"Li, C., Zhou, A., Yao, A.: Omni-dimensional dynamic convolution. arXiv preprint arXiv:.07947 (2022). https:\/\/doi.org\/10.48550\/arXiv.2209.07947","DOI":"10.48550\/arXiv.2209.07947"},{"issue":"9","key":"2060_CR10","doi-asserted-by":"publisher","first-page":"299","DOI":"10.1016\/j.cja.2023.03.048","volume":"36","author":"X Yuanliang","year":"2023","unstructured":"Yuanliang, X., Guodong, J., Tao, S., et al.: Template-guided frequency attention and adaptive cross-entropy loss for UAV visual tracking. Chin. J. Aeronaut. 36(9), 299\u2013312 (2023). https:\/\/doi.org\/10.1016\/j.cja.2023.03.048","journal-title":"Chin. J. Aeronaut."},{"key":"2060_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2023.3305728","volume":"61","author":"Y Xue","year":"2023","unstructured":"Xue, Y., Jin, G., Shen, T., et al.: Smalltrack: wavelet pooling and graph enhanced classification for uav small object tracking. IEEE Trans. Geosci. Remote Sens. 61, 1\u201315 (2023). https:\/\/doi.org\/10.1109\/TGRS.2023.3305728","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"2060_CR12","first-page":"51094","volume":"36","author":"C Wang","year":"2023","unstructured":"Wang, C., He, W., Nie, Y., et al.: Gold-YOLO: efficient object detector via gather-and-distribute mechanism. Adv. Neural. Inf. Process. Syst. 36, 51094\u201351112 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2060_CR13","doi-asserted-by":"publisher","first-page":"4341","DOI":"10.1109\/TIP.2023.3297408","volume":"32","author":"Y Quan","year":"2023","unstructured":"Quan, Y., Zhang, D., Zhang, L., et al.: Centralized feature pyramid for object detection. IEEE Trans. Image Process. 32, 4341\u20134354 (2023). https:\/\/doi.org\/10.1109\/TIP.2023.3297408","journal-title":"IEEE Trans. Image Process."},{"issue":"8","key":"2060_CR14","doi-asserted-by":"publisher","first-page":"526","DOI":"10.3390\/drones7080526","volume":"7","author":"Z Zhang","year":"2023","unstructured":"Zhang, Z.: Drone-YOLO: An efficient neural network method for target detection in drone images. Drones. 7(8), 526 (2023). https:\/\/doi.org\/10.3390\/drones7080526","journal-title":"Drones"},{"key":"2060_CR15","doi-asserted-by":"publisher","first-page":"377","DOI":"10.1016\/j.neucom.2022.03.033","volume":"489","author":"R Zhang","year":"2022","unstructured":"Zhang, R., Shao, Z., Huang, X., et al.: Adaptive dense pyramid network for object detection in UAV imagery. Neurocomputing. 489, 377\u2013389 (2022). https:\/\/doi.org\/10.1016\/j.neucom.2022.03.033","journal-title":"Neurocomputing"},{"issue":"2","key":"2060_CR16","doi-asserted-by":"publisher","first-page":"268","DOI":"10.1016\/j.ejrs.2024.03.001","volume":"27","author":"Y Chen","year":"2024","unstructured":"Chen, Y., Liu, Z., Zhang, L., et al.: MFFNet: A lightweight multi-feature fusion network for UAV infrared object detection. Egypt. J. Remote Sens. 27(2), 268\u2013276 (2024). https:\/\/doi.org\/10.1016\/j.ejrs.2024.03.001","journal-title":"Egypt. J. Remote Sens."},{"key":"2060_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2024.3363057","volume":"62","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., Ye, M., Zhu, G., et al.: FFCA-YOLO for small object detection in remote sensing images. IEEE Trans. Geosci. Remote Sens. 62, 1\u201315 (2024). https:\/\/doi.org\/10.1109\/TGRS.2024.3363057","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"2060_CR18","doi-asserted-by":"publisher","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. Adv. Neural. Inf. Process. Syst. 30 (2017). https:\/\/doi.org\/10.48550\/arXiv.1706.03762","DOI":"10.48550\/arXiv.1706.03762"},{"key":"2060_CR19","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"2060_CR20","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., et al.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229 Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2060_CR21","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.04159","author":"X Zhu","year":"2020","unstructured":"Zhu, X., Su, W., Lu, L., et al.: Deformable detr: deformable transformers for end-to-end object detection. ArXiv Preprint arXiv:2010 04159. (2020). https:\/\/doi.org\/10.48550\/arXiv.2010.04159","journal-title":"ArXiv Preprint arXiv:2010 04159"},{"key":"2060_CR22","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Liu, S., et al.: Dn-detr: accelerate detr training by introducing query denoising. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13619\u201313627 (2022)","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"2060_CR23","doi-asserted-by":"crossref","unstructured":"Chen, Q., Chen, X., Wang, J., et al.: Group detr: fast detr training with group-wise one-to-many assignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6633\u20136642 (2023)","DOI":"10.1109\/ICCV51070.2023.00610"},{"key":"2060_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, M., Song, G., Liu, Y., et al.: Decoupled detr: Spatially disentangling localization and classification for improved end-to-end object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6601\u20136610 (2023)","DOI":"10.1109\/ICCV51070.2023.00607"},{"key":"2060_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2023.3314641","volume":"61","author":"H Wu","year":"2023","unstructured":"Wu, H., Huang, P., Zhang, M., et al.: CMTFNet: CNN and multiscale transformer fusion network for remote-sensing image semantic segmentation. IEEE Trans. Geosci. Remote Sens. 61, 1\u201312 (2023). https:\/\/doi.org\/10.1109\/TGRS.2023.3314641","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"2060_CR26","doi-asserted-by":"crossref","unstructured":"Li, J., Wen, Y., He, L.: Scconv: Spatial and channel reconstruction convolution for feature redundancy. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6153\u20136162 (2023)","DOI":"10.1109\/CVPR52729.2023.00596"},{"key":"2060_CR27","doi-asserted-by":"crossref","unstructured":"Shi, D.: Transnext: robust foveal visual perception for vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17773\u201317783 (2024)","DOI":"10.1109\/CVPR52733.2024.01683"},{"key":"2060_CR28","doi-asserted-by":"crossref","unstructured":"Cui, Y., Ren, W., Knoll, A.: Omni-kernel network for image restoration. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 1426\u20131434 (2024)","DOI":"10.1609\/aaai.v38i2.27907"},{"key":"2060_CR29","doi-asserted-by":"publisher","first-page":"101870","DOI":"10.1016\/j.inffus.2023.101870","volume":"99","author":"L Tang","year":"2023","unstructured":"Tang, L., Zhang, H., Xu, H., et al.: Rethinking the necessity of image fusion in high-level vision tasks: a practical infrared and visible image fusion network based on progressive semantic injection and scene fidelity. Inf. Fusion. 99, 101870 (2023). https:\/\/doi.org\/10.1016\/j.inffus.2023.101870","journal-title":"Inf. Fusion"},{"key":"2060_CR30","doi-asserted-by":"crossref","unstructured":"Rahman, M.M., Munir, M., Marculescu, R.: Emcad: efficient multi-scale convolutional attention decoding for medical image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11769\u201311779 (2024)","DOI":"10.1109\/CVPR52733.2024.01118"},{"key":"2060_CR31","first-page":"107984","volume":"37","author":"A Wang","year":"2024","unstructured":"Wang, A., Chen, H., Liu, L., et al.: Yolov10: real-time end-to-end object detection. Adv. Neural. Inf. Process. Syst. 37, 107984\u2013108011 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2060_CR32","doi-asserted-by":"crossref","unstructured":"Wu, B., Wan, A., Yue, X., et al.: Shift: A zero flop, zero parameter alternative to spatial convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9127\u20139135 (2018)","DOI":"10.1109\/CVPR.2018.00951"},{"key":"2060_CR33","unstructured":"Zhang, T., Li, L., Zhou, Y., et al.: Cas-vit: Convolutional additive self-attention vision Transformers for efficient mobile applications. ArXiv Preprint arXiv:240803703 (2024). https:\/\/doi.org\/10.48550\/arXiv.2408.03703"},{"key":"2060_CR34","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2307.07662","author":"S Ma","year":"2023","unstructured":"Ma, S., Xu, Y.: Mpdiou: a loss for efficient and accurate bounding box regression. ArXiv Preprint arXiv:2307.07662. (2023). https:\/\/doi.org\/10.48550\/arXiv.2307.07662","journal-title":"ArXiv Preprint arXiv"},{"key":"2060_CR35","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2401.10525","author":"H Zhang","year":"2024","unstructured":"Zhang, H., Zhang, S.: Focaler-iou: more focused intersection over union loss. ArXiv Preprint arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2401.10525 :2401.10525","journal-title":"ArXiv Preprint arXiv"},{"issue":"11","key":"2060_CR36","doi-asserted-by":"publisher","first-page":"7380","DOI":"10.1109\/TPAMI.2021.3119563","volume":"44","author":"P Zhu","year":"2021","unstructured":"Zhu, P., Wen, L., Du, D., et al.: Detection and tracking meet drones challenge. IEEE Trans. Pattern Anal. Mach. Intell. 44(11), 7380\u20137399 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2021.3119563","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"2060_CR37","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1038\/s41597-023-02066-6","volume":"10","author":"J Suo","year":"2023","unstructured":"Suo, J., Wang, T., Zhang, X., et al.: HIT-UAV: a high-altitude infrared thermal dataset for unmanned aerial vehicle-based object detection. Sci. Data. 10(1), 227 (2023). https:\/\/doi.org\/10.1038\/s41597-023-02066-6","journal-title":"Sci. Data"},{"issue":"6","key":"2060_CR38","doi-asserted-by":"publisher","first-page":"3329","DOI":"10.1007\/s00530-023-01182-y","volume":"29","author":"X Wang","year":"2023","unstructured":"Wang, X., He, N., Hong, C., et al.: Yolo-erf: Lightweight object detector for uav aerial images. Multimedia Syst. 29(6), 3329\u20133339 (2023)","journal-title":"Multimedia Syst."},{"key":"2060_CR39","doi-asserted-by":"crossref","unstructured":"Ding, X., Zhang, X., Han, J., et al.: Scaling up your kernels to 31x31: revisiting large kernel design in cnns. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11963\u201311975 (2022)","DOI":"10.1109\/CVPR52688.2022.01166"},{"key":"2060_CR40","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2304.03198","author":"X Zhang","year":"2023","unstructured":"Zhang, X., Liu, C., Yang, D., et al.: RFAConv: innovating spatial attention and standard convolutional operation. ArXiv Preprint arXiv:2304 03198. (2023). https:\/\/doi.org\/10.48550\/arXiv.2304.03198","journal-title":"ArXiv Preprint arXiv:2304 03198"},{"issue":"11","key":"2060_CR41","doi-asserted-by":"publisher","first-page":"9528","DOI":"10.1109\/TNNLS.2022.3151138","volume":"34","author":"J Zhong","year":"2022","unstructured":"Zhong, J., Chen, J., Mian, A.: DualConv: dual convolutional kernels for lightweight deep neural networks. IEEE Trans. Neural Netw. Learn. Syst. 34(11), 9528\u20139535 (2022). https:\/\/doi.org\/10.1109\/TNNLS.2022.3151138","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"2060_CR42","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Chen, D., et al.: Cswin transformer: A general vision transformer backbone with cross-shaped windows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12124\u201312134 (2022)","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"2060_CR43","doi-asserted-by":"crossref","unstructured":"Ding, X., Zhang, X., Han, J., et al.: Diverse branch block: Building a convolution as an inception-like unit. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10886\u201310895 (2021)","DOI":"10.1109\/CVPR46437.2021.01074"},{"key":"2060_CR44","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2311.11587","author":"X Zhang","year":"2023","unstructured":"Zhang, X., Song, Y., Song, T., et al.: AKConv: Convolutional kernel with arbitrary sampled shapes and arbitrary number of parameters. ArXiv Preprint arXiv. (2023). https:\/\/doi.org\/10.48550\/arXiv.2311.11587 :2311.11587","journal-title":"ArXiv Preprint arXiv"},{"key":"2060_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, J., Li, X., Li, J., et al.: Rethinking mobile block for efficient attention-based models. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1389\u20131400 IEEE Computer Society (2023)","DOI":"10.1109\/ICCV51070.2023.00134"},{"key":"2060_CR46","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., et al.: Vision transformer with deformable attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4794\u20134803 (2022)","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"2060_CR47","doi-asserted-by":"crossref","unstructured":"Sun, S., Ren, W., Gao, X., et al.: Restoring images in adverse weather conditions via histogram transformer. In: European Conference on Computer Vision, pp. 111\u2013129 Springer (2024)","DOI":"10.1007\/978-3-031-72670-5_7"},{"key":"2060_CR48","doi-asserted-by":"crossref","unstructured":"Shaker, A., Maaz, M., Rasheed, H., et al.: Swiftformer: Efficient additive attention for transformer-based real-time mobile vision applications. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17425\u201317436 (2023)","DOI":"10.1109\/ICCV51070.2023.01598"},{"key":"2060_CR49","first-page":"14541","volume":"35","author":"Z Pan","year":"2022","unstructured":"Pan, Z., Cai, J., Zhuang, B.: Fast vision Transformers with Hilo attention. Adv. Neural. Inf. Process. Syst. 35, 14541\u201314554 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2060_CR50","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2405.11582","author":"J Guo","year":"2024","unstructured":"Guo, J., Chen, X., Tang, Y., et al.: Slab: Efficient Transformers with simplified linear attention and progressive re-parameterized batch normalization. ArXiv Preprint arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2405.11582 :2405.11582","journal-title":"ArXiv Preprint arXiv"},{"issue":"6","key":"2060_CR51","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster R-CNN: Towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016). https:\/\/doi.org\/10.1109\/TPAMI.2016.2577031","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2060_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, S., Chi, C., Yao, Y., et al.: Bridging the gap between anchor-based and anchor-free detection via adaptive training sample selection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9759\u20139768 (2020)","DOI":"10.1109\/CVPR42600.2020.00978"},{"key":"2060_CR53","doi-asserted-by":"crossref","unstructured":"Feng, C., Zhong, Y., Gao, Y., et al.: Tood: Task-aligned one-stage object detection. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3490\u20133499 IEEE Computer Society (2021)","DOI":"10.1109\/ICCV48922.2021.00349"},{"key":"2060_CR54","unstructured":"Li, X., Wang, W., Wu, L., et al.: Generalized focal loss: Learning qualified and distributed bounding boxes for dense object detection. Adv. Neural Inf. Process. Syst. 33, 21002\u201321012 (2020)"},{"key":"2060_CR55","doi-asserted-by":"crossref","unstructured":"Sohan, M., Ram, S., Reddy, T.R.: C. V.: A review on yolov8 and its advancements. In: International Conference on Data Intelligence and Cognitive Informatics, pp. 529\u2013545 Springer (2024)","DOI":"10.1007\/978-981-99-7962-2_39"},{"key":"2060_CR56","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2410.17725","author":"R Khanam","year":"2024","unstructured":"Khanam, R., Hussain, M.: Yolov11: An overview of the key architectural enhancements. ArXiv Preprint arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2410.17725 :2410.17725","journal-title":"ArXiv Preprint arXiv"},{"key":"2060_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Zou, H., He, S., et al.: YOLOX-Drone: an improved object detection method for UAV images. In: IGARSS 2024\u20132024 IEEE International Geoscience and Remote Sensing Symposium, pp. 7676\u20137680 IEEE (2024)","DOI":"10.1109\/IGARSS53475.2024.10642459"},{"issue":"3","key":"2060_CR58","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1007\/s10044-024-01323-7","volume":"27","author":"T Ning","year":"2024","unstructured":"Ning, T., Wu, W., Zhang, J.: Small object detection based on YOLOv8 in UAV perspective. Pattern Anal. Appl. 27(3), 103 (2024). https:\/\/doi.org\/10.1007\/s10044-024-01323-7","journal-title":"Pattern Anal. Appl."},{"key":"2060_CR59","doi-asserted-by":"crossref","unstructured":"Yang, L., Wang, Y., Kong, L., et al.: Object detection of visdrone based on attention mechanism and FasterNet. In: 2024 5th International Conference on Computer Vision, Image and Deep Learning (CVIDL), pp. 257\u2013261 IEEE (2024)","DOI":"10.1109\/CVIDL62147.2024.10603457"},{"key":"2060_CR60","doi-asserted-by":"crossref","unstructured":"Du, B., Huang, Y., Chen, J., et al.: Adaptive sparse convolutional networks with global context enhancement for faster object detection on drone images. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 13435\u201313444 (2023)","DOI":"10.1109\/CVPR52729.2023.01291"},{"issue":"14","key":"2060_CR61","doi-asserted-by":"publisher","first-page":"3468","DOI":"10.3390\/rs15143468","volume":"15","author":"L Zhou","year":"2023","unstructured":"Zhou, L., Liu, Z., Zhao, H.: A multi-scale object detector based on coordinate and global information aggregation for UAV aerial images. Remote Sens. 15(14), 3468 (2023). https:\/\/doi.org\/10.3390\/rs15143468","journal-title":"Remote Sens."},{"issue":"3","key":"2060_CR62","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1007\/s00530-024-01342-8","volume":"30","author":"F Sun","year":"2024","unstructured":"Sun, F., He, N., Li, R., et al.: GD-PAN: A multiscale fusion architecture applied to object detection in UAV aerial images. Multimedia Syst. 30(3), 143 (2024)","journal-title":"Multimedia Syst."},{"key":"2060_CR63","doi-asserted-by":"crossref","unstructured":"Du, D., Qi, Y., Yu, H., et al.: The unmanned aerial vehicle benchmark: object detection and tracking. In: Proceedings of the European conference on computer vision (ECCV), pp. 370\u2013386 (2018)","DOI":"10.1007\/978-3-030-01249-6_23"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02060-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-02060-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02060-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T04:21:18Z","timestamp":1770783678000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-02060-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,4]]},"references-count":63,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2060"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-02060-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,4]]},"assertion":[{"value":"3 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"14"}}