{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:59:41Z","timestamp":1780973981125,"version":"3.54.1"},"reference-count":80,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61472220"],"award-info":[{"award-number":["61472220"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61572286"],"award-info":[{"award-number":["61572286"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132117","type":"journal-article","created":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T17:48:34Z","timestamp":1773942514000},"page":"132117","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Semantic and spatial feature reinforcement for object detection"],"prefix":"10.1016","volume":"319","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6928-3421","authenticated-orcid":false,"given":"Zhenyi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6710-9436","authenticated-orcid":false,"given":"Tianping","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"9","key":"10.1016\/j.eswa.2026.132117_b0005","doi-asserted-by":"crossref","first-page":"7347","DOI":"10.1016\/j.jksuci.2021.08.001","article-title":"A study on generic object detection with emphasis on future research directions","volume":"34","author":"Arulprakash","year":"2022","journal-title":"Journal of King Saud University - Computer and Information Sciences"},{"issue":"5","key":"10.1016\/j.eswa.2026.132117_b0010","doi-asserted-by":"crossref","first-page":"1483","DOI":"10.1109\/TPAMI.2019.2956516","article-title":"Cascade R-CNN: High Quality Object Detection and Instance Segmentation","volume":"43","author":"Cai","year":"2021","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"5","key":"10.1016\/j.eswa.2026.132117_b0015","doi-asserted-by":"crossref","first-page":"2425","DOI":"10.1109\/TNNLS.2021.3106641","article-title":"Hierarchical Regression and Classification for Accurate Object Detection","volume":"34","author":"Cao","year":"2023","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.132117_b0020","first-page":"1971","article-title":"GCNet: Non-Local Networks Meet Squeeze-Excitation Networks and beyond","volume":"2019","author":"Cao","year":"2019","journal-title":"IEEE\/CVF International Conference on Computer Vision Workshop (ICCVW)"},{"key":"10.1016\/j.eswa.2026.132117_b0025","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-End Object Detection with Transformers. In A. Vedaldi, H. Bischof, T. Brox, & J.-M. Frahm (Eds.), Computer Vision \u2013 ECCV 2020 (Vol. 12346, pp. 213\u2013229). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"10.1016\/j.eswa.2026.132117_b0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.120539","article-title":"HA-Transformer: Harmonious aggregation from local to global for object detection","volume":"230","author":"Chen","year":"2023","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132117_b0035","doi-asserted-by":"crossref","first-page":"1002","DOI":"10.1109\/TIP.2024.3354108","article-title":"DEA-Net: Single image Dehazing based on Detail-Enhanced Convolution and Content-Guided attention","volume":"33","author":"Chen","year":"2024","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132117_b0040","doi-asserted-by":"crossref","first-page":"284","DOI":"10.1109\/TMM.2023.3264008","article-title":"DDOD: Dive deeper into the Disentanglement of Object Detector","volume":"26","author":"Chen","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132117_b0045","first-page":"7369","article-title":"Dynamic Head: Unifying Object Detection Heads with Attentions","volume":"2021","author":"Dai","year":"2021","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"issue":"2","key":"10.1016\/j.eswa.2026.132117_b0050","doi-asserted-by":"crossref","first-page":"85","DOI":"10.1007\/s13748-019-00203-0","article-title":"Convolutional neural network: A review of models, methodologies and applications to object detection","volume":"9","author":"Dhillon","year":"2020","journal-title":"Progress in Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0055","doi-asserted-by":"crossref","DOI":"10.1016\/j.cosrev.2024.100686","article-title":"A survey of deep learning techniques for detecting and recognizing objects in complex environments","volume":"54","author":"Dogra","year":"2024","journal-title":"Computer Science Review"},{"issue":"1","key":"10.1016\/j.eswa.2026.132117_b0060","doi-asserted-by":"crossref","first-page":"534","DOI":"10.1609\/aaai.v36i1.19932","article-title":"Construct Effective Geometry Aware Feature Pyramid Network for Multi-Scale Object Detection","volume":"36","author":"Dong","year":"2022","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0065","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., & Houlsby, N. (2021). An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale (arXiv:2010.11929). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2010.11929."},{"key":"10.1016\/j.eswa.2026.132117_b0070","first-page":"213","article-title":"VisDrone-DET2019: The Vision Meets Drone Object Detection in image Challenge results","volume":"2019","author":"Du","year":"2019","journal-title":"IEEE\/CVF International Conference on Computer Vision Workshop (ICCVW)"},{"issue":"9","key":"10.1016\/j.eswa.2026.132117_b0075","doi-asserted-by":"crossref","first-page":"277","DOI":"10.1007\/s10462-025-11284-w","article-title":"Comprehensive review of recent developments in visual object detection based on deep learning","volume":"58","author":"Edozie","year":"2025","journal-title":"Artificial Intelligence Review"},{"issue":"2","key":"10.1016\/j.eswa.2026.132117_b0080","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","article-title":"The Pascal Visual Object classes (VOC) Challenge","volume":"88","author":"Everingham","year":"2010","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132117_b0085","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2021.108437","article-title":"Adaptive region-aware feature enhancement for object detection","volume":"124","author":"Fan","year":"2022","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132117_b0090","first-page":"3490","article-title":"TOOD: Task-aligned One-stage Object Detection","volume":"2021","author":"Feng","year":"2021","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"10.1016\/j.eswa.2026.132117_b0095","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.109376","article-title":"LKASR: Large kernel attention for lightweight image super-resolution","volume":"252","author":"Feng","year":"2022","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132117_b0100","unstructured":"Fu, L., Tian, H., Zhai, X. B., Gao, P., & Peng, X. (2022). IncepFormer: Efficient Inception Transformer with Pyramid Pooling for Semantic Segmentation (arXiv:2212.03035). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2212.03035."},{"issue":"8","key":"10.1016\/j.eswa.2026.132117_b0105","doi-asserted-by":"crossref","first-page":"3588","DOI":"10.1109\/TNNLS.2020.3015790","article-title":"Topology of Learning in Feedforward Neural Networks","volume":"32","author":"Gabella","year":"2021","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.132117_b0110","unstructured":"Guo, M.-H., Lu, C.-Z., Hou, Q., Liu, Z., Cheng, M.-M., & Hu, S.-M. (2022). SegNeXt: Rethinking Convolutional Attention Design for Semantic Segmentation (arXiv:2209.08575). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2209.08575."},{"issue":"1","key":"10.1016\/j.eswa.2026.132117_b0115","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A Survey on Vision Transformer","volume":"45","author":"Han","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"8","key":"10.1016\/j.eswa.2026.132117_b0120","doi-asserted-by":"crossref","first-page":"9306","DOI":"10.1109\/TPAMI.2023.3238011","article-title":"Adaptive Feature selection with Augmented Attributes","volume":"45","author":"Hou","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0125","doi-asserted-by":"crossref","unstructured":"Hou, X., Liu, M., Zhang, S., Wei, P., Chen, B., & Lan, X. (2025). Relation DETR: Exploring Explicit Position Relation Prior for Object Detection. In A. Leonardis, E. Ricci, S. Roth, O. Russakovsky, T. Sattler, & G. Varol (Eds.), Computer Vision \u2013 ECCV 2024 (Vol. 15108, pp. 89\u2013105). Springer Nature Switzerland. https:\/\/doi.org\/10.1007\/978-3-031-72973-7_6.","DOI":"10.1007\/978-3-031-72973-7_6"},{"key":"10.1016\/j.eswa.2026.132117_b0130","first-page":"15338","article-title":"A2 -FPN: Attention Aggregation based Feature Pyramid Network for Instance Segmentation","volume":"2021","author":"Hu","year":"2021","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0135","doi-asserted-by":"crossref","first-page":"8906","DOI":"10.1109\/TMM.2023.3243616","article-title":"DilateFormer: Multi-Scale Dilated Transformer for Visual Recognition","volume":"25","author":"Jiao","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.132117_b0140","doi-asserted-by":"crossref","first-page":"9445","DOI":"10.1109\/TIP.2020.3028196","article-title":"SAFNet: A Semi-Anchor-Free Network with Enhanced Feature Pyramid for Object Detection","volume":"29","author":"Jin","year":"2020","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132117_b0145","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Liu, S., Guo, J., Ni, L. M., & Zhang, L. (2022). DN-DETR: Accelerate DETR Training by Introducing Query DeNoising (arXiv:2203.01305). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2203.01305.","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"10.1016\/j.eswa.2026.132117_b0150","unstructured":"Li, J., Xia, X., Li, W., Li, H., Wang, X., Xiao, X., Wang, R., Zheng, M., & Pan, X. (2022). Next-ViT: Next Generation Vision Transformer for Efficient Deployment in Realistic Industrial Scenarios (arXiv:2207.05501). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2207.05501."},{"issue":"10","key":"10.1016\/j.eswa.2026.132117_b0155","doi-asserted-by":"crossref","first-page":"12581","DOI":"10.1109\/TPAMI.2023.3282631","article-title":"UniFormer: Unifying Convolution and Self-attention for Visual Recognition","volume":"45","author":"Li","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0160","first-page":"6843","article-title":"AlignDet: Aligning Pre-training and Fine-tuning in Object Detection","volume":"2023","author":"Li","year":"2023","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"10.1016\/j.eswa.2026.132117_b0165","first-page":"510","article-title":"Selective Kernel Networks","volume":"2019","author":"Li","year":"2019","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0170","first-page":"16748","article-title":"Large Selective Kernel Network for Remote Sensing Object Detection","volume":"2023","author":"Li","year":"2023","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"issue":"6","key":"10.1016\/j.eswa.2026.132117_b0175","doi-asserted-by":"crossref","first-page":"2683","DOI":"10.1109\/TCSVT.2022.3218880","article-title":"Dense Crosstalk Feature Aggregation for Classification and Localization in Object Detection","volume":"33","author":"Li","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132117_b0180","series-title":"2019 IEEE 31st International Conference on Tools with Artificial Intelligence (ICTAI)","first-page":"1702","article-title":"TFPN: Twin Feature Pyramid Networks for Object Detection","author":"Liang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132117_b0185","first-page":"936","article-title":"Feature Pyramid Networks for Object Detection","volume":"2017","author":"Lin","year":"2017","journal-title":"IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0190","first-page":"2999","article-title":"Focal loss for Dense Object Detection","volume":"2017","author":"Lin","year":"2017","journal-title":"IEEE International Conference on Computer Vision (ICCV)"},{"key":"10.1016\/j.eswa.2026.132117_b0195","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Bourdev, L., Girshick, R., Hays, J., Perona, P., Ramanan, D., Zitnick, C. L., & Doll\u00e1r, P. (2015). Microsoft COCO: Common Objects in Context (arXiv:1405.0312). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1405.0312."},{"key":"10.1016\/j.eswa.2026.132117_b0200","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109878","article-title":"Feature disentanglement in one-stage object detection","volume":"145","author":"Lin","year":"2024","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132117_b0205","doi-asserted-by":"crossref","first-page":"2746","DOI":"10.1109\/TIP.2024.3378457","article-title":"CCDet: Confidence-Consistent Learning for Dense Object Detection","volume":"33","author":"Liu","year":"2024","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.132117_b0210","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2025.113081","article-title":"Rethinking the multi-scale feature hierarchy in object detection transformer (DETR)","volume":"175","author":"Liu","year":"2025","journal-title":"Applied Soft Computing"},{"key":"10.1016\/j.eswa.2026.132117_b0215","unstructured":"Liu, S., Li, F., Zhang, H., Yang, X., Qi, X., Su, H., Zhu, J., & Zhang, L. (2022). DAB-DETR: Dynamic Anchor Boxes are Better Queries for DETR (arXiv:2201.12329). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2201.12329."},{"key":"10.1016\/j.eswa.2026.132117_b0220","first-page":"8759","article-title":"Path Aggregation Network for Instance Segmentation","volume":"2018","author":"Liu","year":"2018","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132117_b0225","first-page":"15539","article-title":"SAP-DETR: Bridging the Gap between Salient Points and Queries-based Transformer Detector for Fast Model Convergency","volume":"2023","author":"Liu","year":"2023","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0230","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Zheng, H.-T., & Sun, J. (2018). ShuffleNet V2: Practical Guidelines for Efficient CNN Architecture Design. In V. Ferrari, M. Hebert, C. Sminchisescu, & Y. Weiss (Eds.), Computer Vision \u2013 ECCV 2018 (Vol. 11218, pp. 122\u2013138). Springer International Publishing. https:\/\/doi.org\/10.1007\/978-3-030-01264-9_8.","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"10.1016\/j.eswa.2026.132117_b0235","doi-asserted-by":"crossref","DOI":"10.1016\/j.jestch.2025.102161","article-title":"A comprehensive review on YOLO versions for object detection","volume":"70","author":"Murat","year":"2025","journal-title":"Engineering Science and Technology, an International Journal"},{"key":"10.1016\/j.eswa.2026.132117_b0240","doi-asserted-by":"crossref","unstructured":"Nan, Z., Li, X., Dai, J., & Xiang, T. (2025). MI-DETR: An Object Detection Model with Multi-time Inquiries Mechanism (arXiv:2503.01463). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2503.01463.","DOI":"10.1109\/CVPR52734.2025.00443"},{"key":"10.1016\/j.eswa.2026.132117_b0245","unstructured":"Narayanan, M. (2023). SENetV2: Aggregated dense layer for channelwise and global representations (arXiv:2311.10807). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2311.10807."},{"key":"10.1016\/j.eswa.2026.132117_b0250","doi-asserted-by":"crossref","unstructured":"Park, H.-J., Choi, Y.-J., Lee, Y.-W., & Kim, B.-G. (2022). ssFPN: Scale Sequence (S^2) Feature Based-Feature Pyramid Network for Object Detection (arXiv:2208.11533). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2208.11533.","DOI":"10.3390\/s23094432"},{"issue":"6","key":"10.1016\/j.eswa.2026.132117_b0255","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks","volume":"39","author":"Ren","year":"2017","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0260","first-page":"658","article-title":"Generalized Intersection over Union: A Metric and a loss for Bounding Box Regression","volume":"2019","author":"Rezatofighi","year":"2019","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0265","first-page":"11560","article-title":"Revisiting the Sibling Head in Object Detector","volume":"2020","author":"Song","year":"2020","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0270","first-page":"14449","article-title":"Sparse R-CNN: End-to-End Object Detection with Learnable proposals","volume":"2021","author":"Sun","year":"2021","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0275","first-page":"3591","article-title":"Rethinking Transformer-based Set Prediction for Object Detection","volume":"2021","author":"Sun","year":"2021","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"issue":"5","key":"10.1016\/j.eswa.2026.132117_b0280","doi-asserted-by":"crossref","first-page":"3383","DOI":"10.1109\/TCSVT.2023.3323879","article-title":"A Refinement Method for Single-Stage Object Detection based on Progressive Decoupled Task Alignment","volume":"34","author":"Tang","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132117_b0285","first-page":"9626","article-title":"FCOS: Fully Convolutional One-Stage Object Detection","volume":"2019","author":"Tian","year":"2019","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"issue":"4","key":"10.1016\/j.eswa.2026.132117_b0290","doi-asserted-by":"crossref","first-page":"5314","DOI":"10.1109\/TPAMI.2022.3206148","article-title":"ResMLP: Feedforward Networks for image Classification with Data-Efficient Training","volume":"45","author":"Touvron","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0295","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L., & Polosukhin, I. (2023). Attention Is All You Need (arXiv:1706.03762). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1706.03762."},{"key":"10.1016\/j.eswa.2026.132117_b0300","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2022.117682","article-title":"SLMS-SSD: Improving the balance of semantic and spatial information in object detection","volume":"206","author":"Wang","year":"2022","journal-title":"Expert Systems with Applications"},{"issue":"3","key":"10.1016\/j.eswa.2026.132117_b0305","doi-asserted-by":"crossref","first-page":"2567","DOI":"10.1609\/aaai.v36i3.20158","article-title":"Anchor DETR: Query Design for Transformer-Based Detector","volume":"36","author":"Wang","year":"2022","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"issue":"9","key":"10.1016\/j.eswa.2026.132117_b0310","doi-asserted-by":"crossref","first-page":"16760","DOI":"10.1109\/TNNLS.2025.3562588","article-title":"Bridging Task-specific and Task-Interactive Features with Opportune Branching and Adaptive attention for Object Detection","volume":"36","author":"Wen","year":"2025","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.132117_b0315","first-page":"10183","article-title":"Rethinking Classification and Localization for Object Detection","volume":"2020","author":"Wu","year":"2020","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0320","doi-asserted-by":"crossref","first-page":"58","DOI":"10.1016\/j.neucom.2022.09.132","article-title":"Rethinking prediction alignment in one-stage object detection","volume":"514","author":"Xiao","year":"2022","journal-title":"Neurocomputing"},{"issue":"12","key":"10.1016\/j.eswa.2026.132117_b0325","doi-asserted-by":"crossref","first-page":"15171","DOI":"10.1109\/TPAMI.2023.3319634","article-title":"Mutual-Assistance Learning for Object Detection","volume":"45","author":"Xie","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132117_b0330","first-page":"1","article-title":"ASSD: Feature Aligned Single-Shot Detection for Multiscale Objects in Aerial Imagery","volume":"60","author":"Xu","year":"2022","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"10.1016\/j.eswa.2026.132117_b0335","doi-asserted-by":"crossref","DOI":"10.1016\/j.compbiomed.2023.107149","article-title":"EFPN: Effective medical image detection using feature pyramid fusion enhancement","volume":"163","author":"Xu","year":"2023","journal-title":"Computers in Biology and Medicine"},{"key":"10.1016\/j.eswa.2026.132117_b0340","doi-asserted-by":"crossref","unstructured":"Xue, Z., Chen, W., & Li, J. (2020). Enhancement and Fusion of Multi-Scale Feature Maps for Small Object Detection. 2020 39th Chinese Control Conference (CCC), 7212\u20137217. https:\/\/doi.org\/10.23919\/CCC50068.2020.9189352.","DOI":"10.23919\/CCC50068.2020.9189352"},{"issue":"9","key":"10.1016\/j.eswa.2026.132117_b0345","doi-asserted-by":"crossref","first-page":"7820","DOI":"10.1109\/TCSVT.2024.3376773","article-title":"Asymptotic Feature Pyramid Network for labeling Pixels and Regions","volume":"34","author":"Yang","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132117_b0350","first-page":"5672","article-title":"InceptionNeXt: When Inception Meets ConvNeXt","volume":"2024","author":"Yu","year":"2024","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"issue":"8","key":"10.1016\/j.eswa.2026.132117_b0355","doi-asserted-by":"crossref","first-page":"2825","DOI":"10.1007\/s11263-024-02005-x","article-title":"Semantic-Aligned Matching for Enhanced DETR Convergence and Multi-Scale Feature Fusion","volume":"132","author":"Zhang","year":"2024","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.132117_b0360","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L. M., & Shum, H.-Y. (2022). DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection (arXiv:2203.03605). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2203.03605."},{"key":"10.1016\/j.eswa.2026.132117_b0365","doi-asserted-by":"crossref","unstructured":"Zhang, Q.-L., & Yang, Y.-B. (2021). SA-Net: Shuffle Attention for Deep Convolutional Neural Networks. ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 2235\u20132239. https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9414568.","DOI":"10.1109\/ICASSP39728.2021.9414568"},{"key":"10.1016\/j.eswa.2026.132117_b0370","first-page":"9756","article-title":"Bridging the Gap between Anchor-based and Anchor-Free Detection via Adaptive Training Sample selection","volume":"2020","author":"Zhang","year":"2020","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0375","first-page":"12073","article-title":"TopFormer: Token Pyramid Transformer for Mobile Semantic Segmentation","volume":"2022","author":"Zhang","year":"2022","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0380","first-page":"9397","article-title":"Localization Distillation for Dense Object Detection","volume":"2022","author":"Zheng","year":"2022","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0385","first-page":"13062","article-title":"Squeeze-and-attention Networks for Semantic Segmentation","volume":"2020","author":"Zhong","year":"2020","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"10.1016\/j.eswa.2026.132117_b0390","first-page":"3735","article-title":"Orientation robust object detection in aerial images using deep convolutional neural network","volume":"2015","author":"Zhu","year":"2015","journal-title":"IEEE International Conference on Image Processing (ICIP)"},{"key":"10.1016\/j.eswa.2026.132117_b0395","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2021). Deformable DETR: Deformable Transformers for End-to-End Object Detection (arXiv:2010.04159). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2010.04159."},{"key":"10.1016\/j.eswa.2026.132117_b0400","first-page":"6725","article-title":"DETRs with Collaborative Hybrid Assignments Training","volume":"2023","author":"Zong","year":"2023","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010304?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010304?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:34:23Z","timestamp":1780972463000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426010304"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":80,"alternative-id":["S0957417426010304"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132117","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic and spatial feature reinforcement for object detection","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132117","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132117"}}