{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:03:54Z","timestamp":1784531034473,"version":"3.55.0"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100005645","name":"Socialist Republic of Vietnam Ministry of Education and Training","doi-asserted-by":"publisher","award":["CT2025.EA.BKA.09"],"award-info":[{"award-number":["CT2025.EA.BKA.09"]}],"id":[{"id":"10.13039\/501100005645","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.imavis.2026.106097","type":"journal-article","created":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T16:34:18Z","timestamp":1782491658000},"page":"106097","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["AQS-IDETR: Adaptive Query Selection for Efficient Inference in Real-Time Detection Transformers"],"prefix":"10.1016","volume":"173","author":[{"given":"Nguyen Quoc Nhat","family":"Minh","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nguyen Hong","family":"Dang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ngo Van","family":"Linh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dinh Viet","family":"Sang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9856-2110","authenticated-orcid":false,"given":"Duc Anh","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.106097_b1","doi-asserted-by":"crossref","unstructured":"R. Girshick, J. Donahue, T. Darrell, J. Malik, Rich feature hierarchies for accurate object detection and semantic segmentation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2014, pp. 580\u2013587.","DOI":"10.1109\/CVPR.2014.81"},{"key":"10.1016\/j.imavis.2026.106097_b2","doi-asserted-by":"crossref","unstructured":"R. Girshick, Fast r-cnn, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 1440\u20131448.","DOI":"10.1109\/ICCV.2015.169"},{"key":"10.1016\/j.imavis.2026.106097_b3","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","volume":"28","author":"Ren","year":"2015","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.106097_b4","article-title":"R-fcn: Object detection via region-based fully convolutional networks","volume":"29","author":"Dai","year":"2016","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.106097_b5","doi-asserted-by":"crossref","unstructured":"J. Redmon, S. Divvala, R. Girshick, A. Farhadi, You only look once: Unified, real-time object detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 779\u2013788.","DOI":"10.1109\/CVPR.2016.91"},{"key":"10.1016\/j.imavis.2026.106097_b6","doi-asserted-by":"crossref","unstructured":"J. Redmon, A. Farhadi, YOLO9000: better, faster, stronger, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 7263\u20137271.","DOI":"10.1109\/CVPR.2017.690"},{"key":"10.1016\/j.imavis.2026.106097_b7","series-title":"YOLOv5 by ultralytics","author":"Jocher","year":"2020"},{"key":"10.1016\/j.imavis.2026.106097_b8","series-title":"YOLOX: Exceeding YOLO series in 2021","author":"Ge","year":"2021"},{"key":"10.1016\/j.imavis.2026.106097_b9","series-title":"YOLOv6: A single-stage object detection framework for industrial applications","author":"Li","year":"2022"},{"key":"10.1016\/j.imavis.2026.106097_b10","doi-asserted-by":"crossref","unstructured":"C.-Y. Wang, A. Bochkovskiy, H.-Y.M. Liao, YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 7464\u20137475.","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"10.1016\/j.imavis.2026.106097_b11","series-title":"Ultralytics YOLO","author":"Jocher","year":"2023"},{"key":"10.1016\/j.imavis.2026.106097_b12","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Yolov9: Learning what you want to learn using programmable gradient information","author":"Wang","year":"2024"},{"key":"10.1016\/j.imavis.2026.106097_b13","doi-asserted-by":"crossref","first-page":"107984","DOI":"10.52202\/079017-3429","article-title":"Yolov10: Real-time end-to-end object detection","volume":"37","author":"Wang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.106097_b14","series-title":"YOLOv12: Attention-centric real-time object detectors","author":"Tian","year":"2025"},{"key":"10.1016\/j.imavis.2026.106097_b15","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.imavis.2026.106097_b16","series-title":"Deformable detr: Deformable transformers for end-to-end object detection","author":"Zhu","year":"2020"},{"key":"10.1016\/j.imavis.2026.106097_b17","series-title":"Efficient DETR: Improving end-to-end object detector with dense prior","author":"Yao","year":"2021"},{"key":"10.1016\/j.imavis.2026.106097_b18","series-title":"Dab-detr: Dynamic anchor boxes are better queries for detr","author":"Liu","year":"2022"},{"key":"10.1016\/j.imavis.2026.106097_b19","doi-asserted-by":"crossref","unstructured":"F. Li, H. Zhang, S. Liu, J. Guo, L.M. Ni, L. Zhang, Dndetr: Accelerate detr training by introducing query denoising, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 13619\u201313627.","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"10.1016\/j.imavis.2026.106097_b20","series-title":"Dino: Detr with improved denoising anchor boxes for end-to-end object detection","author":"Zhang","year":"2022"},{"key":"10.1016\/j.imavis.2026.106097_b21","doi-asserted-by":"crossref","unstructured":"Q. Chen, X. Chen, J. Wang, S. Zhang, K. Yao, H. Feng, J. Han, E. Ding, G. Zeng, J. Wang, Group detr: Fast detr training with group-wise one-to-many assignment, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 6633\u20136642.","DOI":"10.1109\/ICCV51070.2023.00610"},{"key":"10.1016\/j.imavis.2026.106097_b22","doi-asserted-by":"crossref","unstructured":"Y. Wang, X. Li, S. Weng, G. Zhang, H. Yue, H. Feng, J. Han, E. Ding, KD-DETR: Knowledge Distillation for Detection Transformer with Consistent Distillation Points Sampling, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 16016\u201316025.","DOI":"10.1109\/CVPR52733.2024.01516"},{"key":"10.1016\/j.imavis.2026.106097_b23","doi-asserted-by":"crossref","unstructured":"Z. Zong, G. Song, Y. Liu, Detrs with collaborative hybrid assignments training, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 6748\u20136758.","DOI":"10.1109\/ICCV51070.2023.00621"},{"key":"10.1016\/j.imavis.2026.106097_b24","doi-asserted-by":"crossref","unstructured":"Y. Zhao, W. Lv, S. Xu, J. Wei, G. Wang, Q. Dang, Y. Liu, J. Chen, DETRs Beat YOLOs on Real-time Object Detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 16965\u201316974.","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"10.1016\/j.imavis.2026.106097_b25","series-title":"RT-DETRv2: Improved baseline with bag-of-freebies for real-time detection transformer","author":"Lv","year":"2024"},{"key":"10.1016\/j.imavis.2026.106097_b26","doi-asserted-by":"crossref","unstructured":"S. Wang, C. Xia, F. Lv, Y. Shi, RT-DETRv3: Real-Time End-to-End Object Detection with Hierarchical Dense Positive Supervision, in: Proceedings of the Winter Conference on Applications of Computer Vision, WACV, 2025, pp. 1628\u20131636.","DOI":"10.1109\/WACV61041.2025.00166"},{"key":"10.1016\/j.imavis.2026.106097_b27","doi-asserted-by":"crossref","unstructured":"Q. Zhou, C. Yu, Z. Wang, F. Wang, D2Q-DETR: Decoupling and Dynamic Queries for Oriented Object Detection with Transformers, in: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP, 2023, pp. 1\u20135.","DOI":"10.1109\/ICASSP49357.2023.10095341"},{"key":"10.1016\/j.imavis.2026.106097_b28","series-title":"European Conference on Computer Vision","first-page":"290","article-title":"DQ-DETR: DETR with dynamic query for tiny object detection","author":"Huang","year":"2025"},{"issue":"14","key":"10.1016\/j.imavis.2026.106097_b29","doi-asserted-by":"crossref","DOI":"10.3390\/app15147686","article-title":"Layer-wise query selection to eliminate redundant queries in DETR","volume":"15","author":"Hong","year":"2025","journal-title":"Appl. Sci."},{"key":"10.1016\/j.imavis.2026.106097_b30","series-title":"Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.imavis.2026.106097_b31","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep Residual Learning for Image Recognition, in: 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.imavis.2026.106097_b32","doi-asserted-by":"crossref","unstructured":"T. He, Z. Zhang, H. Zhang, Z. Zhang, J. Xie, M. Li, Bag of Tricks for Image Classification with Convolutional Neural Networks, in: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2018, pp. 558\u2013567.","DOI":"10.1109\/CVPR.2019.00065"},{"key":"10.1016\/j.imavis.2026.106097_b33","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, B. Guo, Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.imavis.2026.106097_b34","series-title":"PolypDB: A curated multi-center dataset for development of ai algorithms in colonoscopy","author":"Jha","year":"2024"},{"issue":"3","key":"10.1016\/j.imavis.2026.106097_b35","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1007\/s42979-023-01748-7","article-title":"Cppe-5: Medical personal protective equipment dataset","volume":"4","author":"Dagli","year":"2023","journal-title":"SN Comput. Sci."}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626002040?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626002040?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T06:24:24Z","timestamp":1784528664000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0262885626002040"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":35,"alternative-id":["S0262885626002040"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.106097","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"AQS-IDETR: Adaptive Query Selection for Efficient Inference in Real-Time Detection Transformers","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.106097","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"106097"}}