{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,19]],"date-time":"2026-08-19T22:58:38Z","timestamp":1787180318578,"version":"build-2736575974"},"reference-count":72,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,2,22]],"date-time":"2025-02-22T00:00:00Z","timestamp":1740182400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,22]],"date-time":"2025-02-22T00:00:00Z","timestamp":1740182400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s00371-025-03825-9","type":"journal-article","created":{"date-parts":[[2025,2,22]],"date-time":"2025-02-22T07:52:00Z","timestamp":1740210720000},"page":"7585-7601","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["DMTNet: dual-domain adaptive multi-scale feature fusion network with transformer for small target detection"],"prefix":"10.1007","volume":"41","author":[{"given":"Yan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xueting","family":"Sang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yemei","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shudong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengpei","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,2,22]]},"reference":[{"key":"3825_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patter.2024.100929","author":"Bo Qian","year":"2024","unstructured":"Qian, Bo., et al.: DRAC 2022: A public benchmark for diabetic retinopathy analysis on ultra-wide optical coherence tomography angiography images. Patterns (2024). https:\/\/doi.org\/10.1016\/j.patter.2024.100929","journal-title":"Patterns"},{"issue":"1","key":"3825_CR2","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1109\/TII.2021.3085669","volume":"18","author":"J Li","year":"2021","unstructured":"Li, J., Chen, J., Sheng, B., Li, P., Yang, P., Feng, D.D., Qi, J.: Automatic detection and classification system of domestic waste via multimodel cascaded convolutional neural network. IEEE Trans. Ind. Inf. 18(1), 163\u2013173 (2021)","journal-title":"IEEE Trans. Ind. Inf."},{"key":"3825_CR3","unstructured":"Yiming, C., Yan, L.,  Cao, Z., Liu,  D.: Tf-blender: temporal feature blender for video object detection.\" In: Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 8138\u20138147(2021)"},{"key":"3825_CR4","doi-asserted-by":"crossref","unstructured":"Huang, C., Wu Z., Wen J., Xu Y., Jiang Q., Wang Y.: Abnormal event detection using deep contrastive learning for intelligent video surveillance system. IEEE Trans. Ind. Inform. 18(8), 5171\u20135179 (2021)","DOI":"10.1109\/TII.2021.3122801"},{"key":"3825_CR5","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/j.isprsjprs.2017.11.011","volume":"140","author":"N Audebert","year":"2018","unstructured":"Audebert, N., Le Saux, B., Lef\u00e8vre, S.: Beyond RGB: very high resolution urban remote sensing with multimodal deep networks. ISPRS J. Photogramm. Remote Sens. 140, 20\u201332 (2018)","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"3825_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.comcom.2019.10.007","volume":"149","author":"S Sudhakar","year":"2020","unstructured":"Sudhakar, S., Vijayakumar, V., Kumar, C.S., Priya, V., Ravi, L., Subramaniyaswamy, V.: Unmanned aerial vehicle (UAV) based forest fire detection and monitoring for reducing false alarms in forest-fires. Computer Commun. 149, 1\u201316 (2020)","journal-title":"Computer Commun."},{"key":"3825_CR7","doi-asserted-by":"publisher","first-page":"100","DOI":"10.3390\/rs9020100","volume":"9","author":"MB Bejiga","year":"2017","unstructured":"Bejiga, M.B., Zeggada, A., Nouffidj, A., Melgani, F.: A convolutional neural network approach for assisting avalanche search and rescue operations with UAV imagery. Remote Sens. 9, 100 (2017)","journal-title":"Remote Sens."},{"key":"3825_CR8","unstructured":"Wang, Wenguan, Cheng Han, Tianfei Zhou, and Dongfang Liu. \"Visual recognition with deep nearest centroids.\" arXiv preprint arXiv:2209.07383 (2022)."},{"key":"3825_CR9","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R. and Farhadi, A.: You only look once: Unified, real-time object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 779\u2013788 (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"3825_CR10","doi-asserted-by":"crossref","unstructured":"Wei, L., Dragomir, A., Dumitru, E., Christian, S., Scott, R., Yang, F., and etc, SSD: Single shot multibox detector. In: European con- ference on computer vision. Springer, Cham, pp. 21\u201337 (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"3825_CR11","doi-asserted-by":"crossref","unstructured":"Tan, M., Pang, R., Le, Q.: EfficientDet: scalable and efficient object detection. In: Proceedings of the IEEE\/CVF conference on com- puter vision and pattern recognition, pp. 10781\u201310790 (2020)","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"3825_CR12","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T. and Malik, J.: Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 580\u2013587 (2014)","DOI":"10.1109\/CVPR.2014.81"},{"key":"3825_CR13","unstructured":"Ren, S., He, K., Girshick, R., and Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. Adv. Neural Inf. Process. Syst. (2015)"},{"key":"3825_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Dollar, P., and Girshick, R.: Mask R-CNN. In: Proceedings of the IEEE International Conference on Computer vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"3825_CR15","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. (2017)"},{"key":"3825_CR16","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. and Uszkoreit, J.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"3825_CR17","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S. and Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"issue":"14","key":"3825_CR18","doi-asserted-by":"publisher","first-page":"8161","DOI":"10.3390\/app13148161","volume":"13","author":"M Ma","year":"2023","unstructured":"Ma, M., Pang, H.: SP-YOLOv8s: an improved YOLOv8s model for remote sensing image tiny object detection. Appl. Sci. 13(14), 8161 (2023)","journal-title":"Appl. Sci."},{"key":"3825_CR19","doi-asserted-by":"crossref","unstructured":"Sunkara, R. and Luo, T.: No more strided convolutions or pooling: A new CNN building block for low-resolution images and small objects. In: Joint European Conference on Machine Learning and Knowledge Discovery in Databases, pp. 443\u2013459 (2022)","DOI":"10.1007\/978-3-031-26409-2_27"},{"key":"3825_CR20","unstructured":"G. Jocher. (Jun. 2020). YOLOv5. [Online]. Available: https:\/\/github.com\/ultralytics\/yolov5"},{"key":"3825_CR21","doi-asserted-by":"crossref","unstructured":"Vaswani, A., Ramachandran, P., Srinivas, A., Parmar, N., Hechtman, B. and Shlens, J.: Scaling local self-attention for parameter efficient visual backbones. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12894\u201312904 (2021)","DOI":"10.1109\/CVPR46437.2021.01270"},{"key":"3825_CR22","doi-asserted-by":"crossref","unstructured":"Chen, Y., Dai, X., Liu, M., Chen, D., Yuan, L. and Liu, Z.: Dynamic convolution: Attention over convolution kernels. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 11030\u201311039 (2020)","DOI":"10.1109\/CVPR42600.2020.01104"},{"key":"3825_CR23","doi-asserted-by":"crossref","unstructured":"Liu, Dongfang, Yiming Cui, Wenbo Tan, and Yingjie Chen. \"Sg-net: Spatial granularity network for one-stage video instance segmentation.\" In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 9816\u20139825. 2021.","DOI":"10.1109\/CVPR46437.2021.00969"},{"key":"3825_CR24","doi-asserted-by":"crossref","unstructured":"D Liu Y Cui L Yan C Mousas B Yang Y Chen 2021 Densernet: Weakly supervised visual localization using multi-scale feature aggregation In: Proceedings of the AAAI conference on artificial intelligence 35, 7. pp. 6101 6109","DOI":"10.1609\/aaai.v35i7.16760"},{"key":"3825_CR25","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B. and Belongie, S.: Feature pyramid networks for object detection. In Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"3825_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Lu, X., Cao, G., Yang, Y., Jiao, L. and Liu, F.: ViT-YOLO: Transformer-based YOLO for object detection. In Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 2799\u20132808 (2021)","DOI":"10.1109\/ICCVW54120.2021.00314"},{"issue":"10","key":"3825_CR27","doi-asserted-by":"publisher","first-page":"7853","DOI":"10.1007\/s00521-022-08077-5","volume":"35","author":"J Wang","year":"2023","unstructured":"Wang, J., Chen, Y., Dong, Z., Gao, M.: Improved YOLOv5 network for real-time multi-scale traffic sign detection. Neural Comput. Appl. 35(10), 7853\u20137865 (2023)","journal-title":"Neural Comput. Appl."},{"key":"3825_CR28","doi-asserted-by":"publisher","first-page":"7192","DOI":"10.1109\/TIP.2020.2999854","volume":"29","author":"A Nazir","year":"2020","unstructured":"Nazir, A., Cheema, M.N., Sheng, B., Li, H., Li, P., Yang, P., Jung, Y., Qin, J., Kim, J., Feng, D.D.: OFF-eNET: An optimally fused fully end-to-end network for automatic dense volumetric 3D intracranial blood vessels segmentation. IEEE Transa. Image Proc. 29, 7192\u20137202 (2020)","journal-title":"IEEE Transa. Image Proc."},{"key":"3825_CR29","doi-asserted-by":"crossref","unstructured":"Wang, C., Wei, X. and Jiang, X.: June. MA-YOLO: Multi-Scale Information Prediction Network Based on the Multi-Direction Weighted Pyramid for UAV Scene. In 2023 International Joint Conference on Neural Networks (IJCNN), pp. 01\u201308 (2023)","DOI":"10.1109\/IJCNN54540.2023.10191601"},{"key":"3825_CR30","doi-asserted-by":"crossref","unstructured":"Zhu, X., Lyu, S., Wang, X. and Zhao, Q.: TPH-YOLOv5: Improved YOLOv5 based on transformer prediction head for object detection on drone-captured scenarios. In Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 2778\u20132788 (2021)","DOI":"10.1109\/ICCVW54120.2021.00312"},{"key":"3825_CR31","doi-asserted-by":"crossref","unstructured":"Chen, X., Wang, X., Zhou, J., Qiao, Y. and Dong, C.: Activating more pixels in image super-resolution transformer. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 22367\u201322377 (2023)","DOI":"10.1109\/CVPR52729.2023.02142"},{"key":"3825_CR32","doi-asserted-by":"crossref","unstructured":"Zhu, L., Wang, X., Ke, Z., Zhang, W. and Lau, R.W.: Biformer: Vision transformer with bi-level routing attention. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10323\u201310333 (2023)","DOI":"10.1109\/CVPR52729.2023.00995"},{"key":"3825_CR33","doi-asserted-by":"crossref","unstructured":"Wang, K., Liew, J.H., Zou, Y., Zhou, D. and Feng, J.: Panet: Few-shot image semantic segmentation with prototype alignment. In proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 9197\u20139206 (2019)","DOI":"10.1109\/ICCV.2019.00929"},{"key":"3825_CR34","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Sheng, T., Wang, Y., Tang, Z., Chen, Y., Cai, L. and Ling, H.: July. M2det: A single-shot object detector based on multi-level feature pyramid network. In Proceedings of the AAAI conference on artificial intelligence, pp. 9259\u20139266 (2019)","DOI":"10.1609\/aaai.v33i01.33019259"},{"key":"3825_CR35","doi-asserted-by":"crossref","unstructured":"Yang, G., Lei, J., Zhu, Z., Cheng, S., Feng, Z. and Liang, R.: AFPN: asymptotic feature pyramid network for object detection. In 2023 IEEE International Conference on Systems, Man, and Cybernetics (SMC), pp. 2184\u20132189 (2023)","DOI":"10.1109\/SMC53992.2023.10394415"},{"key":"3825_CR36","doi-asserted-by":"publisher","first-page":"9445","DOI":"10.1109\/TIP.2020.3028196","volume":"29","author":"Z Jin","year":"2020","unstructured":"Jin, Z., Liu, B., Chu, Q., Yu, N.: SAFNet: a semi-anchor-free network with enhanced feature pyramid for object detection. IEEE Transa. Image Proc. 29, 9445\u20139457 (2020)","journal-title":"IEEE Transa. Image Proc."},{"key":"3825_CR37","unstructured":"Wang, C., Wang, H. and Pan, P.: Local contrast and global contextual information make infrared small object salient again. arXiv preprint arXiv:2301.12093 (2023)"},{"key":"3825_CR38","doi-asserted-by":"crossref","unstructured":"Lingyun, G., Popov, E. and Ge, D.: Spectral network combining fourier transformation and deep learning for remote sensing object detection. In 2022 International Conference on Electrical Engineering and Photonics (EExPolytech), pp. 99\u2013102 (2022)","DOI":"10.1109\/EExPolytech56308.2022.9950863"},{"key":"3825_CR39","doi-asserted-by":"crossref","unstructured":"Gao, Ning, Xingyu Jiang, Xiuhui Zhang, and Yue Deng. \"Efficient Frequency-Domain Image Deraining with Contrastive Regularization.\" In European Conference on Computer Vision. Springer, Cham, 2025","DOI":"10.1007\/978-3-031-72940-9_14"},{"key":"3825_CR40","doi-asserted-by":"crossref","unstructured":"Zhou, Man, Jie Huang, Keyu Yan, Danfeng Hong, Xiuping Jia, Jocelyn Chanussot, and Chongyi Li. \"A general spatial-frequency learning framework for multimodal image fusion.\" IEEE Transactions on Pattern Analysis and Machine Intelligence (2024).","DOI":"10.1109\/TPAMI.2024.3368112"},{"key":"3825_CR41","doi-asserted-by":"crossref","unstructured":"Jiang, Xingyu, Xiuhui Zhang, Ning Gao, and Yue Deng. \"When Fast Fourier Transform Meets Transformer for Image Restoration.\" In European Conference on Computer Vision. Springer, Cham, 2025.","DOI":"10.1007\/978-3-031-72995-9_22"},{"key":"3825_CR42","doi-asserted-by":"crossref","unstructured":"Cong, Xiaofeng, Jie Gui, Jing Zhang, Junming Hou, and Hao Shen. \"A Semi-supervised Nighttime Dehazing Baseline with Spatial-Frequency Aware and Realistic Brightness Constraint.\" In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2631\u20132640. 2024.","DOI":"10.1109\/CVPR52733.2024.00254"},{"key":"3825_CR43","doi-asserted-by":"crossref","unstructured":"Suvorov, R., Logacheva, E., Mashikhin, A., Remizova, A., Ashukha, A., Silvestrov, A., Kong, N., Goka, H., Park, K. and Lempitsky, V.: Resolution-robust large mask inpainting with fourier convolutions. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp. 2149\u20132159 (2022)","DOI":"10.1109\/WACV51458.2022.00323"},{"key":"3825_CR44","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A. and Zagoruyko, S.: End-to-end object detection with transformers. In European conference on computer vision, pp. 213\u2013229 (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"3825_CR45","doi-asserted-by":"crossref","unstructured":"Lin, X., Sun, S., Huang, W., Sheng, B., Li, P. and Feng, D.D.:EAPT: efficient attention pyramid transformer for image processing. IEEE Transactions on Multimedia, 25, pp.50\u201361(2021).","DOI":"10.1109\/TMM.2021.3120873"},{"key":"3825_CR46","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.P., Song, K., Liang, D., Lu, T., Luo, P. and Shao, L.: Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In Proceedings of the IEEE\/CVF international conference on computer vision, pp. 568\u2013578 (2021)","DOI":"10.1109\/ICCV48922.2021.00061"},{"issue":"6","key":"3825_CR47","doi-asserted-by":"publisher","first-page":"1687","DOI":"10.3390\/rs15061687","volume":"15","author":"Q Zhao","year":"2023","unstructured":"Zhao, Q., Liu, B., Lyu, S., Wang, C., Zhang, H.: TPH-YOLOv5++: boosting object detection on drone-captured scenarios with cross-layer asymmetric transformer. Remote Sensing 15(6), 1687 (2023)","journal-title":"Remote Sensing"},{"key":"3825_CR48","first-page":"1","volume":"71","author":"Y Dai","year":"2022","unstructured":"Dai, Y., Liu, W., Wang, H., Xie, W., Long, K.: Yolo-former: marrying yolo and transformer for foreign object detection. IEEE Trans. Instrum. Meas. 71, 1\u201314 (2022)","journal-title":"IEEE Trans. Instrum. Meas."},{"issue":"10","key":"3825_CR49","doi-asserted-by":"publisher","first-page":"42851","DOI":"10.1109\/TPAMI.2023.3282631","volume":"45","author":"K Li","year":"2023","unstructured":"Li, K., Wang, Y., Zhang, J., Gao, P., Guanglu Song, Y., Liu, H.L., Qiao, Y.: Uniformer: Unifying convolution and self-attention for visual recognition. IEEE Trans. Pattern Anal. Mach. Intel. 45(10), 42851 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intel."},{"key":"3825_CR50","doi-asserted-by":"crossref","unstructured":"Wu, H., Xiao, B., Codella, N., Liu, M., Dai, X., Yuan, L. and Zhang, L.: Cvt: Introducing convolutions to vision transformers. In Proceedings of the IEEE\/CVF international conference on computer vision, pp. 22\u201331 (2021)","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"3825_CR51","first-page":"30392","volume":"34","author":"T Xiao","year":"2021","unstructured":"Xiao, T., Singh, M., Mintun, E., Darrell, T., Doll\u00e1r, P., Girshick, R.: Early convolutions help transformers see better. Adv. Neural. Inf. Process. Syst. 34, 30392\u201330400 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3825_CR52","doi-asserted-by":"crossref","unstructured":"Yuan, K., Guo, S., Liu, Z., Zhou, A., Yu, F. and Wu, W.: Incorporating convolution designs into visual transformers. In Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 579\u2013588 (2021)","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"3825_CR53","unstructured":"Zhao, Y., Wang, G., Tang, C., Luo, C., Zeng, W. and Zha, Z.J.: A battle of network structures: An empirical study of cnn, transformer, and mlp. arXiv preprint arXiv:2108.13002 (2021)"},{"issue":"11","key":"3825_CR54","doi-asserted-by":"publisher","first-page":"7380","DOI":"10.1109\/TPAMI.2021.3119563","volume":"44","author":"P Zhu","year":"2021","unstructured":"Zhu, P., Wen, L., Du, D., Bian, X., Fan, H., Hu, Q., Ling, H.: Detection and tracking meet drones challenge. IEEE Trans. Pattern Anal. Mach. Intell. 44(11), 7380\u20137399 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3825_CR55","doi-asserted-by":"crossref","unstructured":"Du, D., Qi, Y., Yu, H., Yang, Y., Duan, K., Li, G., Zhang, W., Huang, Q. and Tian, Q.: The unmanned aerial vehicle benchmark: Object detection and tracking. In Proceedings of the European conference on computer vision (ECCV), pp. 370\u2013386 (2018)","DOI":"10.1007\/978-3-030-01249-6_23"},{"key":"3825_CR56","doi-asserted-by":"crossref","unstructured":"Zhu, Zhe, Dun Liang, Songhai Zhang, Xiaolei Huang, Baoli Li, and Shimin Hu. \"Traffic-sign detection and classification in the wild.\" In Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2110\u20132118. 2016.","DOI":"10.1109\/CVPR.2016.232"},{"key":"3825_CR57","doi-asserted-by":"publisher","first-page":"148880","DOI":"10.1109\/ACCESS.2024.3476371","volume":"12","author":"T Wang","year":"2024","unstructured":"Wang, T., Zhang, J., Ren, B., Liu, B.: MMW-YOLOv5: a multi-scale enhanced traffic sign detection algorithm. IEEE Access. 12, 148880 (2024)","journal-title":"IEEE Access."},{"key":"3825_CR58","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K. and Doll\u00e1r, P.: Focal loss for dense object detection. In Proceedings of the IEEE International Conference on Computer vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"3825_CR59","doi-asserted-by":"crossref","unstructured":"Pang, J., Chen, K., Shi, J., Feng, H., Ouyang, W. and Lin, D.: Libra r-cnn: Towards balanced learning for object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 821\u2013830 (2019)","DOI":"10.1109\/CVPR.2019.00091"},{"key":"3825_CR60","doi-asserted-by":"crossref","unstructured":"Wang, C.Y., Bochkovskiy, A. and Liao, H.Y.M.: YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7464\u20137475 (2023)","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"3825_CR61","doi-asserted-by":"publisher","first-page":"103752","DOI":"10.1016\/j.jvcir.2023.103752","volume":"90","author":"M Wang","year":"2023","unstructured":"Wang, M., Yang, W., Wang, L., Chen, D., Wei, F., KeZiErBieKe, H., Liao, Y.: FE-YOLOv5: Feature enhancement network based on YOLOv5 for small object detection. J. Vis. Commun. Image Rep. 90, 103752 (2023)","journal-title":"J. Vis. Commun. Image Rep."},{"key":"3825_CR62","doi-asserted-by":"crossref","unstructured":"Du, B., Huang, Y., Chen, J. and Huang, D.: Adaptive sparse convolutional networks with global context enhancement for faster object detection on drone images. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13435\u201313444 (2023)","DOI":"10.1109\/CVPR52729.2023.01291"},{"key":"3825_CR63","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Lv, W., Xu, S., Wei, J., Wang, G., Dang, Q., Liu, Y. and Chen, J.: Detrs beat yolos on real-time object detection. arXiv preprint arXiv:2304.08069 (2023)","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"3825_CR64","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s12145-024-01265-y","volume":"17","author":"X Cao","year":"2024","unstructured":"Cao, X., Duan, M., Ding, H., Yang, Z.: MS-YOLO: integration-based multi-subnets neural network for object detection in aerial images. Earth Sci. Inf. 17, 1\u201322 (2024)","journal-title":"Earth Sci. Inf."},{"key":"3825_CR65","doi-asserted-by":"crossref","unstructured":"Yang, F., Fan, H., Chu, P., Blasch, E. and Ling, H.: Clustered object detection in aerial images. In Proceedings of the IEEE\/CVF International Conference on Computer vision, pp. 8311\u20138320 (2019)","DOI":"10.1109\/ICCV.2019.00840"},{"key":"3825_CR66","doi-asserted-by":"crossref","unstructured":"Li, C., Yang, T., Zhu, S., Chen, C. and Guan, S.: Density map guided object detection in aerial images. In proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops, pp. 190\u2013191 (2020)","DOI":"10.1109\/CVPRW50498.2020.00103"},{"key":"3825_CR67","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.1109\/TIP.2020.3045636","volume":"30","author":"S Deng","year":"2020","unstructured":"Deng, S., Li, S., Xie, K., Song, W., Liao, X., Hao, A., Qin, H.: A global-local self-adaptive network for drone-view object detection. IEEE Trans. Image Proc. 30, 1556\u20131569 (2020)","journal-title":"IEEE Trans. Image Proc."},{"key":"3825_CR68","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2024.3360408","author":"G Zeng","year":"2024","unstructured":"Zeng, G., Huang, W., Wang, Y., Wang, X., Wenjuan, E.: Transformer fusion and residual learning group classifier loss for long-tailed traffic sign detection. IEEE Sens. J. (2024). https:\/\/doi.org\/10.1109\/JSEN.2024.3360408","journal-title":"IEEE Sens. J."},{"key":"3825_CR69","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3322371","author":"P Liu","year":"2023","unstructured":"Liu, P., Xie, Z., Li, T.: UCN-YOLOv5: Traffic sign target detection algorithm based on deep learning. IEEE Access (2023). https:\/\/doi.org\/10.1109\/ACCESS.2023.3322371","journal-title":"IEEE Access"},{"key":"3825_CR70","doi-asserted-by":"crossref","unstructured":"Li, Pengyu, Chenhe Liu, Tengfei Li, Xinyu Wang, Shihui Zhang, and Dongyang Yu. \"EMDFNet: Efficient Multi-scale and Diverse Feature Network for Traffic Sign Detection.\" In International Conference on Artificial Neural Networks, pp. 120\u2013136. Cham: Springer Nature Switzerland, 2024.","DOI":"10.1007\/978-3-031-72335-3_9"},{"issue":"1","key":"3825_CR71","doi-asserted-by":"publisher","first-page":"25904","DOI":"10.1038\/s41598-024-76804-0","volume":"14","author":"Z Lu","year":"2024","unstructured":"Lu, Z., Zhu, Z., Weipeng, Xu., Li, G., Chen, J.: Enhancing small target traffic sign detection with ML_SAP in YOLOv5s. Sci. Rep. 14(1), 25904 (2024)","journal-title":"Sci. Rep."},{"key":"3825_CR72","doi-asserted-by":"crossref","unstructured":"Yang, C., Huang, Z. and Wang, N.: Querydet: Cascaded sparse query for accelerating high-resolution small object detection. In Proceedings of the IEEE\/CVF Conference on computer vision and pattern recognition, pp. 13668\u201313677 (2022)","DOI":"10.1109\/CVPR52688.2022.01330"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03825-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03825-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03825-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T06:18:50Z","timestamp":1757139530000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03825-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,22]]},"references-count":72,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["3825"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03825-9","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,22]]},"assertion":[{"value":"20 January 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 February 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}