{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,28]],"date-time":"2026-06-28T06:03:46Z","timestamp":1782626626420,"version":"3.54.5"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T00:00:00Z","timestamp":1754179200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T00:00:00Z","timestamp":1754179200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Natural Science Research Project of Anhui Educational Committee","award":["2 024AH040065"],"award-info":[{"award-number":["2 024AH040065"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee","award":["2 024AH040065"],"award-info":[{"award-number":["2 024AH040065"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee","award":["2 024AH040065"],"award-info":[{"award-number":["2 024AH040065"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee","award":["2 024AH040065"],"award-info":[{"award-number":["2 024AH040065"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-07688-8","type":"journal-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T15:33:52Z","timestamp":1754235232000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["FO-YOLO for small object detection in drone aerial imagery"],"prefix":"10.1007","volume":"81","author":[{"given":"Huaping","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kelei","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"7688_CR1","doi-asserted-by":"publisher","first-page":"103910","DOI":"10.1016\/j.imavis.2020.103910","volume":"97","author":"K Tong","year":"2020","unstructured":"Tong K, Wu Y, Zhou F (2020) Recent advances in small object detection based on deep learning: A review. Image Vis Comput 97:103910","journal-title":"Image Vis Comput"},{"issue":"2","key":"7688_CR2","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1109\/MGRS.2019.2902525","volume":"7","author":"M Shimoni","year":"2019","unstructured":"Shimoni M, Haelterman R, Perneel C (2019) Hypersectral imaging for military and security applications: Combining myriad processing and sensing techniques. IEEE Geosci Remote Sens Magaz 7(2):101\u2013117","journal-title":"IEEE Geosci Remote Sens Magaz"},{"key":"7688_CR3","doi-asserted-by":"crossref","unstructured":"Cai Z, Vasconcelos N (2018) Cascade r-cnn: Delving into high quality object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6154\u20136162","DOI":"10.1109\/CVPR.2018.00644"},{"key":"7688_CR4","unstructured":"Ren S (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. arXiv preprint arXiv:1506.01497"},{"key":"7688_CR5","unstructured":"Ross TY, Doll\u00e1r G (2017) Focal loss for dense object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2980\u20132988"},{"key":"7688_CR6","unstructured":"Chen H, He T, Tian Z, et\u00a0al (2020) Fcos: Fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV\u201920)"},{"key":"7688_CR7","unstructured":"Li C, Li L, Jiang H, et\u00a0al (2022) Yolov6: A single-stage object detection framework for industrial applications. arXiv preprint arXiv:2209.02976"},{"key":"7688_CR8","doi-asserted-by":"publisher","first-page":"107455","DOI":"10.1016\/j.engappai.2023.107455","volume":"128","author":"G Song","year":"2024","unstructured":"Song G, Du H, Zhang X et al (2024) Small object detection in unmanned aerial vehicle images using multi-scale hybrid attention. Eng Appl Artif Intell 128:107455","journal-title":"Eng Appl Artif Intell"},{"issue":"1","key":"7688_CR9","doi-asserted-by":"publisher","first-page":"17799","DOI":"10.1038\/s41598-024-68934-2","volume":"14","author":"J Su","year":"2024","unstructured":"Su J, Qin Y, Jia Z et al (2024) Mpe-yolo: enhanced small target detection in aerial imaging. Sci Rep 14(1):17799","journal-title":"Sci Rep"},{"issue":"1","key":"7688_CR10","doi-asserted-by":"publisher","first-page":"134","DOI":"10.3390\/s24010134","volume":"24","author":"J Zhou","year":"2023","unstructured":"Zhou J, Su T, Li K et al (2023) Small target-yolov5: Enhancing the algorithm for small object detection in drone aerial imagery based on yolov5. Sensors 24(1):134","journal-title":"Sensors"},{"key":"7688_CR11","doi-asserted-by":"crossref","unstructured":"Yang J, Zhang X, Song C (2024) Research on a small target object detection method for aerial photography based on improved yolov7. The Visual Computer pp 1\u201315","DOI":"10.1007\/s00371-024-03615-9"},{"issue":"1","key":"7688_CR12","doi-asserted-by":"publisher","first-page":"475","DOI":"10.1109\/TCSVT.2023.3286896","volume":"34","author":"Z Chen","year":"2023","unstructured":"Chen Z, Ji H, Zhang Y et al (2023) High-resolution feature pyramid network for small object detection on drone view. IEEE Trans Circuits Syst Video Technol 34(1):475\u2013489","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"7688_CR13","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"key":"7688_CR14","doi-asserted-by":"crossref","unstructured":"Woo S, Park J, Lee JY, et\u00a0al (2018) Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 3\u201319","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"7688_CR15","doi-asserted-by":"crossref","unstructured":"Wang Q, Wu B, Zhu P, et\u00a0al (2020) Eca-net: Efficient channel attention for deep convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 11534\u201311542","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"7688_CR16","doi-asserted-by":"crossref","unstructured":"Lin TY, Doll\u00e1r P, Girshick R, et\u00a0al (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"7688_CR17","doi-asserted-by":"crossref","unstructured":"Liu S, Qi L, Qin H, et\u00a0al (2018) Path aggregation network for instance segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8759\u20138768","DOI":"10.1109\/CVPR.2018.00913"},{"key":"7688_CR18","doi-asserted-by":"crossref","unstructured":"Tan M, Pang R, Le QV (2020) Efficientdet: Scalable and efficient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10781\u201310790","DOI":"10.1109\/CVPR42600.2020.01079"},{"issue":"6","key":"7688_CR19","doi-asserted-by":"publisher","first-page":"103858","DOI":"10.1016\/j.ipm.2024.103858","volume":"61","author":"R Jing","year":"2024","unstructured":"Jing R, Zhang W, Li Y et al (2024) Dynamic feature focusing network for small object detection. Inf Process Manag 61(6):103858","journal-title":"Inf Process Manag"},{"key":"7688_CR20","doi-asserted-by":"publisher","first-page":"105504","DOI":"10.1016\/j.engappai.2022.105504","volume":"117","author":"QZ Wang Sy","year":"2023","unstructured":"Wang Sy QZ, Cj L et al (2023) Banet: Small and multi-object detection with a bidirectional attention network for traffic scenes. Eng Appl Artif Intell 117:105504","journal-title":"Eng Appl Artif Intell"},{"key":"7688_CR21","doi-asserted-by":"publisher","unstructured":"Jocher G (2020) ultralytics\/yolov5: v3.1 - bug fixes and performance improvements. https:\/\/github.com\/ultralytics\/yolov5, https:\/\/doi.org\/10.5281\/zenodo.4154370","DOI":"10.5281\/zenodo.4154370"},{"issue":"4","key":"7688_CR22","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2017","unstructured":"Chen LC, Papandreou G, Kokkinos I et al (2017) Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans Pattern Anal Mach Intell 40(4):834\u2013848","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7688_CR23","doi-asserted-by":"crossref","unstructured":"Liu S, Huang D, et\u00a0al (2018) Receptive field block net for accurate and fast object detection. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 385\u2013400","DOI":"10.1007\/978-3-030-01252-6_24"},{"key":"7688_CR24","doi-asserted-by":"crossref","unstructured":"Cheng G, Lang C, Wu M, et\u00a0al (2021) Feature enhancement network for object detection in optical remote sensing images. J Remote Sens","DOI":"10.34133\/2021\/9805389"},{"key":"7688_CR25","doi-asserted-by":"publisher","first-page":"103752","DOI":"10.1016\/j.jvcir.2023.103752","volume":"90","author":"M Wang","year":"2023","unstructured":"Wang M, Yang W, Wang L et al (2023) Fe-yolov5: Feature enhancement network based on yolov5 for small object detection. J Vis Commun Image Represent 90:103752","journal-title":"J Vis Commun Image Represent"},{"key":"7688_CR26","doi-asserted-by":"crossref","unstructured":"Ghiasi G, Lin TY, Le QV (2019) Nas-fpn: Learning scalable feature pyramid architecture for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 7036\u20137045","DOI":"10.1109\/CVPR.2019.00720"},{"key":"7688_CR27","doi-asserted-by":"crossref","unstructured":"Guo C, Fan B, Zhang Q, et\u00a0al (2020) Augfpn: Improving multi-scale feature learning for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 12595\u201312604","DOI":"10.1109\/CVPR42600.2020.01261"},{"key":"7688_CR28","doi-asserted-by":"crossref","unstructured":"Zhang YM, Hsieh JW, Lee CC, et\u00a0al (2022) Sfpn: Synthetic fpn for object detection. In: 2022 IEEE International Conference on Image Processing (ICIP), IEEE, pp 1316\u20131320","DOI":"10.1109\/ICIP46576.2022.9897517"},{"key":"7688_CR29","doi-asserted-by":"crossref","unstructured":"Yu J, Jiang Y, Wang Z, et\u00a0al (2016) Unitbox: An advanced object detection network. In: Proceedings of the 24th ACM International Conference on Multimedia, pp 516\u2013520","DOI":"10.1145\/2964284.2967274"},{"key":"7688_CR30","doi-asserted-by":"crossref","unstructured":"Rezatofighi H, Tsoi N, Gwak J, et\u00a0al (2019) Generalized intersection over union: A metric and a loss for bounding box regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 658\u2013666","DOI":"10.1109\/CVPR.2019.00075"},{"key":"7688_CR31","doi-asserted-by":"crossref","unstructured":"Zheng Z, Wang P, Liu W, et\u00a0al (2020) Distance-iou loss: Faster and better learning for bounding box regression. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 12993\u201313000","DOI":"10.1609\/aaai.v34i07.6999"},{"key":"7688_CR32","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1016\/j.neucom.2022.07.042","volume":"506","author":"YF Zhang","year":"2022","unstructured":"Zhang YF, Ren W, Zhang Z et al (2022) Focal and efficient iou loss for accurate bounding box regression. Neurocomputing 506:146\u2013157","journal-title":"Neurocomputing"},{"key":"7688_CR33","unstructured":"Gevorgyan Z (2022) Siou loss: More powerful learning for bounding box regression. arXiv preprint arXiv:2205.12740"},{"key":"7688_CR34","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TGRS.2024.3363057","volume":"62","author":"Y Zhang","year":"2024","unstructured":"Zhang Y, Ye M, Zhu G et al (2024) Ffca-yolo for small object detection in remote sensing images. IEEE Trans Geosci Remote Sens 62:1\u20131. https:\/\/doi.org\/10.1109\/TGRS.2024.3363057","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"7688_CR35","unstructured":"Xu X, Jiang Y, Chen W, et\u00a0al (2022) Damo-yolo: A report on real-time object detection design. arXiv preprint arXiv:2211.15444"},{"key":"7688_CR36","doi-asserted-by":"crossref","unstructured":"Guo J, Ma X, Sansom A, et\u00a0al (2020) Spanet: Spatial pyramid attention network for enhanced image recognition. In: 2020 IEEE International Conference on Multimedia and Expo (ICME), IEEE, pp 1\u20136","DOI":"10.1109\/ICME46284.2020.9102906"},{"key":"7688_CR37","doi-asserted-by":"crossref","unstructured":"Ge Z, Liu S, Li Z, et\u00a0al (2021) Ota: Optimal transport assignment for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 303\u2013312","DOI":"10.1109\/CVPR46437.2021.00037"},{"key":"7688_CR38","unstructured":"Cuturi M (2013) Sinkhorn distances: Lightspeed computation of optimal transport. Adv Neural Inf Process Syst 26"},{"issue":"11","key":"7688_CR39","doi-asserted-by":"publisher","first-page":"7380","DOI":"10.1109\/TPAMI.2021.3119563","volume":"44","author":"P Zhu","year":"2021","unstructured":"Zhu P, Wen L, Du D et al (2021) Detection and tracking meet drones challenge. IEEE Trans Pattern Anal Mach Intell 44(11):7380\u20137399","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7688_CR40","doi-asserted-by":"crossref","unstructured":"Yu X, Gong Y, Jiang N, et\u00a0al (2020) Scale match for tiny person detection. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 1257\u20131265","DOI":"10.1109\/WACV45572.2020.9093394"},{"key":"7688_CR41","unstructured":"Paszke A, Gross S, Massa F, et\u00a0al (2019) Pytorch: An imperative style, high-performance deep learning library. Adv Neural Inf Process Syst 32"},{"key":"7688_CR42","doi-asserted-by":"crossref","unstructured":"Tian Z, Shen C, Chen H, et\u00a0al (2019) Fcos: Fully convolutional one-stage object detection. arxiv 2019. arXiv preprint arXiv:1904.01355","DOI":"10.1109\/ICCV.2019.00972"},{"key":"7688_CR43","unstructured":"Jocher G, Chaurasia A, Qiu J (2023) Ultralytics YOLO. https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"7688_CR44","doi-asserted-by":"crossref","unstructured":"Zhang J, Li X, Li J, et\u00a0al (2023) Rethinking mobile block for efficient attention-based models. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), IEEE Computer Society, pp 1389\u20131400","DOI":"10.1109\/ICCV51070.2023.00134"},{"key":"7688_CR45","doi-asserted-by":"crossref","unstructured":"Yu W, Luo M, Zhou P, et\u00a0al (2022) Metaformer is actually what you need for vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10819\u201310829","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"7688_CR46","doi-asserted-by":"crossref","unstructured":"Yang C, Huang Z, Wang N (2022) Querydet: Cascaded sparse query for accelerating high-resolution small object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 13668\u201313677","DOI":"10.1109\/CVPR52688.2022.01330"},{"key":"7688_CR47","unstructured":"Chen H, Wang Y, Guo J, et\u00a0al (2024) Vanillanet: the power of minimalism in deep learning. Adv Neural Inf Process Syst 36"},{"key":"7688_CR48","first-page":"100523","volume":"27","author":"S Xu","year":"2024","unstructured":"Xu S, Zhang M, Chen J et al (2024) Yolo-hypervision: A vision transformer backbone-based enhancement of yolov5 for detection of dynamic traffic information. Egypt Inf J 27:100523","journal-title":"Egypt Inf J"},{"issue":"5","key":"7688_CR49","doi-asserted-by":"publisher","first-page":"241","DOI":"10.1007\/s00530-024-01447-0","volume":"30","author":"S Peng","year":"2024","unstructured":"Peng S, Fan X, Tian S et al (2024) Ps-yolo: a small object detector based on efficient convolution and multi-scale feature fusion. Multimedia Syst 30(5):241","journal-title":"Multimedia Syst"},{"key":"7688_CR50","unstructured":"Wang A, Chen H, Liu L, et\u00a0al (2024) Yolov10: Real-time end-to-end object detection. arXiv preprint arXiv:2405.14458"},{"key":"7688_CR51","unstructured":"Khanam R, Hussain M (2024) Yolov11: An overview of the key architectural enhancements. arXiv preprint arXiv:2410.17725"},{"key":"7688_CR52","first-page":"1","volume-title":"Computer vision and pattern recognition","author":"A Farhadi","year":"2018","unstructured":"Farhadi A, Redmon J (2018) Yolov3: An incremental improvement. Computer vision and pattern recognition. Springer, Berlin\/Heidelberg, Germany, pp 1\u20136"},{"key":"7688_CR53","unstructured":"Bochkovskiy A, Wang CY, Liao HYM (2020) Yolov4: Optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934"},{"key":"7688_CR54","doi-asserted-by":"crossref","unstructured":"Wang CY, Bochkovskiy A, Liao HYM (2023) Yolov7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 7464\u20137475","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"7688_CR55","doi-asserted-by":"crossref","unstructured":"Wang CY, Yeh IH, Mark\u00a0Liao HY (2025) Yolov9: Learning what you want to learn using programmable gradient information. In: European Conference on Computer Vision, Springer, pp 1\u201321","DOI":"10.1007\/978-3-031-72751-1_1"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07688-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-07688-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-07688-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T13:08:23Z","timestamp":1757336903000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-07688-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":55,"journal-issue":{"issue":"12","published-online":{"date-parts":[[2025,8]]}},"alternative-id":["7688"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-07688-8","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"15 July 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 August 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The data used in this study do not involve ethical experiments.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}}],"article-number":"1208"}}