{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T16:49:45Z","timestamp":1764175785587,"version":"3.44.0"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,3,24]],"date-time":"2025-03-24T00:00:00Z","timestamp":1742774400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,24]],"date-time":"2025-03-24T00:00:00Z","timestamp":1742774400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.61673396"],"award-info":[{"award-number":["No.61673396"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["No.ZR2022MF260"],"award-info":[{"award-number":["No.ZR2022MF260"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s00530-025-01761-1","type":"journal-article","created":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T22:43:09Z","timestamp":1742942589000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Do-DETR: enhancing DETR training convergence with integrated denoising and RoI mechanism"],"prefix":"10.1007","volume":"31","author":[{"given":"Hong","family":"Liang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingwen","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,24]]},"reference":[{"key":"1761_CR1","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778. (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1761_CR2","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 7132\u20137141. (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1761_CR3","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 10\u00a0012\u201310\u00a0022. (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1761_CR4","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition, arXiv preprint arXiv:1409.1556, (2014)"},{"key":"1761_CR5","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z., Sun, Y., He, T., Mueller, J., Manmatha, R. et\u00a0al.: Resnest: Split-attention networks, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 2736\u20132746. (2022)","DOI":"10.1109\/CVPRW56347.2022.00309"},{"issue":"22","key":"1761_CR6","doi-asserted-by":"publisher","first-page":"4530","DOI":"10.3390\/electronics13224530","volume":"13","author":"S Umirzakova","year":"2024","unstructured":"Umirzakova, S., Abdullaev, M., Mardieva, S., Latipova, N., Muksimova, S.: Simplified knowledge distillation for deep neural networks bridging the performance gap with a novel teacher-student architecture. Electronics 13(22), 4530 (2024)","journal-title":"Electronics"},{"key":"1761_CR7","doi-asserted-by":"crossref","unstructured":"Chen, C., Chen, Z., Zhang, J., Tao, D.: Sasa: Semantics-augmented set abstraction for point-based 3d object detection. In: Proceedings of the AAAI conference on artificial intelligence 36(1), 221\u2013229 (2022)","DOI":"10.1609\/aaai.v36i1.19897"},{"key":"1761_CR8","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2117\u20132125. (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"1761_CR9","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: Towards real-time object detection with region proposal networks, Adv. Neural Inf. Process. Syst., 28, (2015)"},{"key":"1761_CR10","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, X., Pang, J., Lin, D.: Fcos3d: Fully convolutional one-stage monocular 3d object detection, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 913\u2013922. (2021)","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"1761_CR11","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G. Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers, In: European conference on computer vision. Springer, pp. 213\u2013229 , (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1761_CR12","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.\u00a0L.: Microsoft coco: common objects in context, In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13. Springer, pp. 740\u2013755 , (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1761_CR13","doi-asserted-by":"crossref","unstructured":"Wang, Y., Zhang, X., Yang, T., Sun, J.: Anchor detr: query design for transformer-based detector, In: Proceedings of the AAAI conference on artificial intelligence, pp. 2567\u20132575. (2022)","DOI":"10.1609\/aaai.v36i3.20158"},{"key":"1761_CR14","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable detr: Deformable transformers for end-to-end object detection, arXiv:2010.04159, (2020)"},{"key":"1761_CR15","doi-asserted-by":"crossref","unstructured":"Sun, Z., Cao, S., Yang, Y., Kitani, K.\u00a0M.: Rethinking transformer-based set prediction for object detection, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 3611\u20133620, (2021)","DOI":"10.1109\/ICCV48922.2021.00359"},{"key":"1761_CR16","unstructured":"Liu, S., Li, F., Zhang, H., Yang, X., Qi, X., Su, H., Zhu, J., Zhang, L.: Dab-detr: Dynamic anchor boxes are better queries for detr, arXiv:2201.12329, (2022)"},{"key":"1761_CR17","doi-asserted-by":"crossref","unstructured":"Dai, X., Chen, Y., Yang, J., Zhang, P., Yuan, L., Zhang, L.: Dynamic detr: End-to-end object detection with dynamic attention, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 2988\u20132997. (2021)","DOI":"10.1109\/ICCV48922.2021.00298"},{"key":"1761_CR18","doi-asserted-by":"crossref","unstructured":"Li, Z., Hu, J., Wu, K., Miao, J., Wu, J.: Adjacent-atrous mechanism for expanding global receptive fields: an end-to-end network for multi-attribute scene analysis in remote sensing imagery, IEEE Trans. Geosci. Remote Sens., (2024)","DOI":"10.1109\/TGRS.2024.3422007"},{"issue":"1","key":"1761_CR19","doi-asserted-by":"publisher","first-page":"12597","DOI":"10.1038\/s41598-024-63363-7","volume":"14","author":"Z Li","year":"2024","unstructured":"Li, Z., Hu, J., Wu, K., Miao, J., Zhao, Z., Wu, J.: Local feature acquisition and global context understanding network for very high-resolution land cover classification. Sci. Rep. 14(1), 12597 (2024)","journal-title":"Sci. Rep."},{"key":"1761_CR20","doi-asserted-by":"crossref","unstructured":"Li, Z., Hu, J., Wu, K., Miao, J., Wu, J.: Comprehensive attribute difference attention network for remote sensing image semantic understanding, IEEE Trans. Geosci. Remote Sens., (2024)","DOI":"10.1109\/TGRS.2024.3516501"},{"key":"1761_CR21","doi-asserted-by":"crossref","unstructured":"Meng, D., Chen, X., Fan, Z., Zeng, G., Li, H., Yuan, Y., Sun, L., Wang, J.: Conditional detr for fast training convergence, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 3651\u20133660. (2021)","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"1761_CR22","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks, Adv. Neural Inf Process. Syst., 25, (2012)"},{"issue":"1","key":"1761_CR23","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1109\/TPAMI.2015.2437384","volume":"38","author":"R Girshick","year":"2015","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Region-based convolutional networks for accurate object detection and segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 38(1), 142\u2013158 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1761_CR24","doi-asserted-by":"publisher","first-page":"84379","DOI":"10.1109\/ACCESS.2023.3302692","volume":"11","author":"J Talreja","year":"2023","unstructured":"Talreja, J., Aramvith, S., Onoye, T.: Dans: deep attention network for single image super-resolution. IEEE Access 11, 84379\u201384397 (2023)","journal-title":"IEEE Access"},{"key":"1761_CR25","doi-asserted-by":"crossref","unstructured":"Talreja, J., Aramvith, S., Onoye, T.: Dhtcun: deep hybrid transformer cnn u network for single-image super-resolution, IEEE Access, (2024)","DOI":"10.1109\/ACCESS.2024.3450300"},{"issue":"2","key":"1761_CR26","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1007\/s40747-024-01760-1","volume":"11","author":"J Talreja","year":"2025","unstructured":"Talreja, J., Aramvith, S., Onoye, T.: Xtnsr: Xception-based transformer network for single image super resolution. Compl. Intell. Syst. 11(2), 162 (2025)","journal-title":"Compl. Intell. Syst."},{"key":"1761_CR27","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: Unified, real-time object detection, In: Proceedings of the IEEE conference on computer vision and pattern recognition, (2016), pp. 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"key":"1761_CR28","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., Berg, A.\u00a0C.: Ssd: Single shot multibox detector, In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14. Springer, pp. 21\u201337. (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"1761_CR29","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast r-cnn, In: Proceedings of the IEEE international conference on computer vision, (2015), pp. 1440\u20131448","DOI":"10.1109\/ICCV.2015.169"},{"key":"1761_CR30","doi-asserted-by":"crossref","unstructured":"Stewart, R., Andriluka, M., Ng, A.\u00a0Y.: End-to-end people detection in crowded scenes, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2325\u20132333. (2016)","DOI":"10.1109\/CVPR.2016.255"},{"key":"1761_CR31","unstructured":"Salvador, A., Bellver, M., Campos, V., Baradad, M., Marques, F., Torres, J., Giro-i\u00a0Nieto, X.: Recurrent neural networks for semantic instance segmentation, arXiv:1712.00617, (2017)"},{"key":"1761_CR32","doi-asserted-by":"crossref","unstructured":"Ren, M., Zemel, R.\u00a0S.: End-to-end instance segmentation with recurrent attention, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 6656\u20136664. (2017)","DOI":"10.1109\/CVPR.2017.39"},{"key":"1761_CR33","unstructured":"Zhang, X., Wan, F., Liu, C., Ji, R., Ye, Q.: Freeanchor: Learning to match anchors for visual object detection, Adv. Neural Inf. Process. Syst, 32, (2019)"},{"key":"1761_CR34","doi-asserted-by":"crossref","unstructured":"Rezatofighi, H., Tsoi, N., Gwak, J., Sadeghian, A., Reid, I., Savarese, S.: Generalized intersection over union: a metric and a loss for bounding box regression, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 658\u2013666. (2019)","DOI":"10.1109\/CVPR.2019.00075"},{"key":"1761_CR35","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Liu, S., Guo, J., Ni, L.\u00a0M., Zhang, L.: Dn-detr: accelerate detr training by introducing query denoising, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 13\u00a0619\u201313\u00a0627. (2022)","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"1761_CR36","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn, In: Proceedings of the IEEE international conference on computer vision, pp. 2961\u20132969. (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"1761_CR37","doi-asserted-by":"crossref","unstructured":"Meng, D., Chen, X., Fan, Z., Zeng, G., Li, H., Yuan, Y., Sun, L., Wang, J.: Conditional detr for fast training convergence, In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 3651\u20133660. (2021)","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"1761_CR38","unstructured":"Yao, Z., Ai, J., Li, B., Zhang, C.: Efficient detr: improving end-to-end object detector with dense prior, arXiv:2104.01318, (2021)"},{"key":"1761_CR39","doi-asserted-by":"crossref","unstructured":"Gao, Z., Wang, L., Han, B., Guo, S.: Adamixer: A fast-converging query-based object detector, In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, (2022), pp. 5364\u20135373","DOI":"10.1109\/CVPR52688.2022.00529"},{"key":"1761_CR40","doi-asserted-by":"crossref","unstructured":"Wang, Y., Zhang, X., Yang, T., Sun, J.: Anchor detr: Query design for transformer-based detector. In: Proceedings of the AAAI conference on artificial intelligence 36(3), 2567\u20132575 (2022)","DOI":"10.1609\/aaai.v36i3.20158"},{"key":"1761_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, G., Luo, Z., Huang, J., Lu, S., Xing, E.\u00a0P.: Semantic-aligned matching for enhanced detr convergence and multi-scale feature fusion, Int. J. Comput. Vis., pp. 1\u201320, (2024)","DOI":"10.1007\/s11263-024-02005-x"},{"key":"1761_CR42","unstructured":"Cai, Z., Liu, S., Wang, G., Ge, Z., Zhang, X., Huang, D.: Align-detr: Improving detr with simple iou-aware bce loss, arXiv:2304.07527, (2023)"},{"key":"1761_CR43","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L.\u00a0M., Shum, H.-Y.: Dino: Detr with improved denoising anchor boxes for end-to-end object detection, arXiv:2203.03605, (2022)"},{"key":"1761_CR44","doi-asserted-by":"crossref","unstructured":"Zheng, D., Dong, W., Hu, H., Chen, X., Wang, Y.: Less is more: Focus attention for efficient detr, In: Proceedings of the IEEE\/CVF international conference on computer vision, (2023), pp. 6674\u20136683","DOI":"10.1109\/ICCV51070.2023.00614"},{"key":"1761_CR45","doi-asserted-by":"crossref","unstructured":"Gao, P., Zheng, M., Wang, X., Dai, J., Li, H.: Fast convergence of detr with spatially modulated co-attention, In: Proceedings of the IEEE\/CVF international conference on computer vision, 2021, pp. 3621\u20133630","DOI":"10.1109\/ICCV48922.2021.00360"},{"key":"1761_CR46","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Lv, W., Xu, S., Wei, J., Wang, G., Dang, Q., Liu, Y., Chen, J.: Detrs beat yolos on real-time object detection,\u201d In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16\u00a0965\u201316\u00a0974. (2024)","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"1761_CR47","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., He, K.: Aggregated residual transformations for deep neural networks, In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1492\u20131500. (2017)","DOI":"10.1109\/CVPR.2017.634"},{"key":"1761_CR48","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.\u00a0N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need, Adv. Neural. Inf. Process. Syst., 30, (2017)"},{"key":"1761_CR49","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection, In: Proceedings of the IEEE international conference on computer vision, pp. 2980\u20132988. (2017)","DOI":"10.1109\/ICCV.2017.324"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01761-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01761-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01761-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,4]],"date-time":"2025-09-04T15:05:18Z","timestamp":1756998318000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01761-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,24]]},"references-count":49,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["1761"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01761-1","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2025,3,24]]},"assertion":[{"value":"5 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}}],"article-number":"171"}}