{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T09:07:23Z","timestamp":1784279243869,"version":"3.55.0"},"reference-count":43,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12101289"],"award-info":[{"award-number":["12101289"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376114"],"award-info":[{"award-number":["62376114"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003392","name":"Fujian Provincial Natural Science Foundation","doi-asserted-by":"publisher","award":["2025J01907"],"award-info":[{"award-number":["2025J01907"]}],"id":[{"id":"10.13039\/501100003392","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007314","name":"Minnan Normal University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100007314","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100020973","name":"Fujian Key Laboratory of Data Science and Statistics","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100020973","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113812","type":"journal-article","created":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T15:22:55Z","timestamp":1776871375000},"page":"113812","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PC","title":["Polarity aware detection transformer with hierarchical cross attention for unmanned aerial vehicle small object detection"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-8005-9345","authenticated-orcid":false,"given":"Huan","family":"Lei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Shang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8372-7314","authenticated-orcid":false,"given":"Wenyuan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113812_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111174","article-title":"Deep learning-enhanced environment perception for autonomous driving: Mdnet with CSP-DarkNet53","volume":"160","author":"Guo","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113812_b2","doi-asserted-by":"crossref","unstructured":"P. Deng, W. Zhou, H. Wu, ChangeChat: An Interactive Model for Remote Sensing Change Analysis via Multimodal Instruction Tuning, in: Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing, 2025, pp. 1\u20135.","DOI":"10.1109\/ICASSP49660.2025.10890620"},{"key":"10.1016\/j.patcog.2026.113812_b3","doi-asserted-by":"crossref","unstructured":"M. Khan, J. Ahmad, A. El Saddik, W. Gueaieb, G. De Masi, F. Karray, Drone-HAT: Hybrid Attention Transformer for Complex Action Recognition in Drone Surveillance Videos, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024, pp. 4713\u20134722.","DOI":"10.1109\/CVPRW63382.2024.00474"},{"key":"10.1016\/j.patcog.2026.113812_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107206","article-title":"An effective method for small object detection in low-resolution images","volume":"127","author":"Jing","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.113812_b5","first-page":"1","article-title":"Yoloow: A spatial scale adaptive real-time object detection neural network for open water search and rescue from UAV aerial imagery","volume":"62","author":"Xu","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113812_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.114052","article-title":"YOLO-ACR: A new architecture for real-time object detection with advanced feature fusion and bounding box regression","volume":"326","author":"Neri","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.patcog.2026.113812_b7","first-page":"1","article-title":"Cross-layer feature pyramid transformer for small object detection in aerial images","volume":"63","author":"Du","year":"2025","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113812_b8","doi-asserted-by":"crossref","first-page":"6437","DOI":"10.1007\/s40747-023-01076-6","article-title":"HRCTNet: A hybrid network with high-resolution representation for object detection in UAV image","volume":"9","author":"Xing","year":"2023","journal-title":"Complex Intell. Syst."},{"key":"10.1016\/j.patcog.2026.113812_b9","first-page":"8673","article-title":"Fbrt-yolo: Faster and better for real-time aerial image detection","volume":"vol. 39","author":"Xiao","year":"2025"},{"key":"10.1016\/j.patcog.2026.113812_b10","doi-asserted-by":"crossref","unstructured":"N. Carion, F. Massa, G. Synnaeve, N. Usunier, A. Kirillov, S. Zagoruyko, End-to-end object detection with transformers, in: Proceedings of the European Conference on Computer Vision, 2020, pp. 213\u2013229.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"10.1016\/j.patcog.2026.113812_b11","doi-asserted-by":"crossref","unstructured":"S. Huang, Z. Lu, X. Cun, Y. Yu, X. Zhou, X. Shen, DEIM: DETR with Improved Matching for Fast Convergence, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2025, pp. 15162\u201315171.","DOI":"10.1109\/CVPR52734.2025.01412"},{"key":"10.1016\/j.patcog.2026.113812_b12","doi-asserted-by":"crossref","unstructured":"Y. Zhao, W. Lv, S. Xu, J. Wei, G. Wang, Q. Dang, Y. Liu, J. Chen, Detrs beat yolos on real-time object detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2024, pp. 16965\u201316974.","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"10.1016\/j.patcog.2026.113812_b13","series-title":"Rt-detrv2: Improved baseline with bag-of-freebies for real-time detection transformer","author":"Lv","year":"2024"},{"key":"10.1016\/j.patcog.2026.113812_b14","unstructured":"Y. Peng, H. Li, P. Wu, Y. Zhang, X. Sun, F. Wu, D-FINE: Redefine Regression Task of DETRs as Fine-grained Distribution Refinement, in: Proceedings of the International Conference on Learning Representations, 2025."},{"key":"10.1016\/j.patcog.2026.113812_b15","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, N. Houlsby, An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale, in: Proceedings of the International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.patcog.2026.113812_b16","doi-asserted-by":"crossref","unstructured":"T.Y. Lin, P. Doll\u00e1r, R. Girshick, K. He, B. Hariharan, S. Belongie, Feature pyramid networks for object detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 2117\u20132125.","DOI":"10.1109\/CVPR.2017.106"},{"key":"10.1016\/j.patcog.2026.113812_b17","doi-asserted-by":"crossref","unstructured":"S. Liu, L. Qi, H. Qin, J. Shi, J. Jia, Path aggregation network for instance segmentation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 8759\u20138768.","DOI":"10.1109\/CVPR.2018.00913"},{"key":"10.1016\/j.patcog.2026.113812_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112955","article-title":"Pcrnet: A multiscale cross-attention network for large deformation medical image registration","volume":"174","author":"Deng","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113812_b19","doi-asserted-by":"crossref","unstructured":"X. Zhang, H. Zeng, S. Guo, L. Zhang, Efficient Long-Range Attention Network for Image Super-Resolution, in: Proceedings of the European Conference on Computer Vision, 2022, pp. 649\u2013667.","DOI":"10.1007\/978-3-031-19790-1_39"},{"key":"10.1016\/j.patcog.2026.113812_b20","doi-asserted-by":"crossref","unstructured":"C.Y. Wang, A. Bochkovskiy, H.Y.M. Liao, YOLOv7: Trainable Bag-of-Freebies Sets New State-of-the-Art for Real-Time Object Detectors, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2023, pp. 7464\u20137475.","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"10.1016\/j.patcog.2026.113812_b21","doi-asserted-by":"crossref","unstructured":"C.Y. Wang, H.Y.M. Liao, Y.H. Wu, P.Y. Chen, J.W. Hsieh, I.H. Yeh, CSPNet: A New Backbone That Can Enhance Learning Capability of CNN, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2020, pp. 1571\u20131580.","DOI":"10.1109\/CVPRW50498.2020.00203"},{"key":"10.1016\/j.patcog.2026.113812_b22","doi-asserted-by":"crossref","unstructured":"C.Y. Wang, I.H. Yeh, H.Y. Mark Liao, YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information, in: Proceedings of the European Conference on Computer Vision, 2025, pp. 1\u201321.","DOI":"10.1007\/978-3-031-72751-1_1"},{"key":"10.1016\/j.patcog.2026.113812_b23","doi-asserted-by":"crossref","first-page":"517","DOI":"10.1016\/j.powtec.2021.04.072","article-title":"Solid particle erosion rate predictions through lsboost","volume":"388","author":"Zhang","year":"2021","journal-title":"Powder Technol."},{"key":"10.1016\/j.patcog.2026.113812_b24","first-page":"22272","article-title":"Dg-mamba: Robust and efficient dynamic graph structure learning with selective state space models","volume":"vol. 39","author":"Yuan","year":"2025"},{"key":"10.1016\/j.patcog.2026.113812_b25","unstructured":"D. Du, P. Zhu, L. Wen, X. Bian, H. Lin, Q. Hu, T. Peng, J. Zheng, X. Wang, Y. Zhang, et al., VisDrone-DET2019: The vision meets drone object detection in image challenge results, in: Proceedings of the IEEE International Conference on Computer Vision, 2019, pp. 213\u2013226."},{"key":"10.1016\/j.patcog.2026.113812_b26","series-title":"HazyDet: Open-source benchmark for drone-view object detection with depth-cues in hazy scenes","author":"Feng","year":"2024"},{"key":"10.1016\/j.patcog.2026.113812_b27","doi-asserted-by":"crossref","first-page":"6700","DOI":"10.1109\/TCSVT.2022.3168279","article-title":"Drone-based rgb-infrared cross-modality vehicle detection via uncertainty-aware learning","volume":"32","author":"Sun","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113812_b28","series-title":"Tinyperson dataset","author":"D.","year":"2023"},{"key":"10.1016\/j.patcog.2026.113812_b29","doi-asserted-by":"crossref","unstructured":"M. Cordts, M. Omran, S. Ramos, T. Rehfeld, M. Enzweiler, R. Benenson, U. Franke, S. Roth, B. Schiele, The cityscapes dataset for semantic urban scene understanding, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 3213\u20133223.","DOI":"10.1109\/CVPR.2016.350"},{"key":"10.1016\/j.patcog.2026.113812_b30","series-title":"Ultralytics YOLOv8","author":"Jocher","year":"2023"},{"key":"10.1016\/j.patcog.2026.113812_b31","doi-asserted-by":"crossref","unstructured":"A. Wang, H. Chen, L. Liu, K. Chen, Z. Lin, J. Han, G. Ding, YOLOv10: real-time end-to-end object detection, in: Proceedings of the International Conference on Neural Information Processing Systems, 2024.","DOI":"10.52202\/079017-3429"},{"key":"10.1016\/j.patcog.2026.113812_b32","series-title":"Ultralytics YOLO11","author":"Jocher","year":"2024"},{"key":"10.1016\/j.patcog.2026.113812_b33","series-title":"YOLOv12: Attention-centric real-time object detectors","author":"Tian","year":"2025"},{"key":"10.1016\/j.patcog.2026.113812_b34","series-title":"YOLOv13: Real-time object detection with hypergraph-enhanced adaptive visual perception","author":"Lei","year":"2025"},{"key":"10.1016\/j.patcog.2026.113812_b35","series-title":"Ultralytics YOLO26","author":"Jocher","year":"2026"},{"key":"10.1016\/j.patcog.2026.113812_b36","first-page":"1","article-title":"CSFPR-RTDETR: Real-time small object detection network for UAV images based on cross-spatial-frequency domain and position relation","volume":"63","author":"Hu","year":"2025","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"10.1016\/j.patcog.2026.113812_b37","doi-asserted-by":"crossref","unstructured":"P. Sun, R. Zhang, Y. Jiang, T. Kong, C. Xu, W. Zhan, M. Tomizuka, L. Li, Z. Yuan, C. Wang, et al., Sparse r-cnn: End-to-end object detection with learnable proposals, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2021, pp. 14454\u201314463.","DOI":"10.1109\/CVPR46437.2021.01422"},{"key":"10.1016\/j.patcog.2026.113812_b38","doi-asserted-by":"crossref","unstructured":"H. Zhang, H. Chang, B. Ma, N. Wang, X. Chen, Dynamic R-CNN: Towards High Quality Object Detection via Dynamic Training, in: Proceedings of the European Conference on Computer Vision, 2020, pp. 260\u2013275.","DOI":"10.1007\/978-3-030-58555-6_16"},{"key":"10.1016\/j.patcog.2026.113812_b39","doi-asserted-by":"crossref","unstructured":"J. Pang, K. Chen, J. Shi, H. Feng, W. Ouyang, D. Lin, Libra R-CNN: Towards Balanced Learning for Object Detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2019, pp. 821\u2013830.","DOI":"10.1109\/CVPR.2019.00091"},{"key":"10.1016\/j.patcog.2026.113812_b40","doi-asserted-by":"crossref","unstructured":"C. Feng, Y. Zhong, Y. Gao, M.R. Scott, W. Huang, TOOD: Task-aligned One-stage Object Detection, in: Proceedings of the IEEE International Conference on Computer Vision, 2021, pp. 3490\u20133499.","DOI":"10.1109\/ICCV48922.2021.00349"},{"key":"10.1016\/j.patcog.2026.113812_b41","doi-asserted-by":"crossref","unstructured":"S. Zhang, X. Wang, J. Wang, J. Pang, C. Lyu, W. Zhang, P. Luo, K. Chen, Dense Distinct Query for End-to-End Object Detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2023, pp. 7329\u20137338.","DOI":"10.1109\/CVPR52729.2023.00708"},{"key":"10.1016\/j.patcog.2026.113812_b42","unstructured":"H. Zhang, F. Li, S. Liu, L. Zhang, H. Su, J. Zhu, L.M. Ni, H. Shum, Dino: Detr with improved denoising anchor boxes for end-to-end object detection, in: Proceedings of the International Conference on Learning Representations, 2023."},{"key":"10.1016\/j.patcog.2026.113812_b43","series-title":"YOLOX: Exceeding YOLO series in 2021","author":"Ge","year":"2021"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007776?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326007776?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T08:38:14Z","timestamp":1784277494000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326007776"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":43,"alternative-id":["S0031320326007776"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113812","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Polarity aware detection transformer with hierarchical cross attention for unmanned aerial vehicle small object detection","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113812","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113812"}}