{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T03:05:10Z","timestamp":1778641510222,"version":"3.51.4"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,2,9]],"date-time":"2024-02-09T00:00:00Z","timestamp":1707436800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,9]],"date-time":"2024-02-09T00:00:00Z","timestamp":1707436800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1007\/s00521-024-09430-6","type":"journal-article","created":{"date-parts":[[2024,2,9]],"date-time":"2024-02-09T19:02:06Z","timestamp":1707505326000},"page":"6793-6808","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["FCDS-DETR: detection transformer based on feature correction and double sampling"],"prefix":"10.1007","volume":"36","author":[{"given":"Min","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiqiang","family":"Jiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhanhua","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shihang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,9]]},"reference":[{"key":"9430_CR1","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1007\/s11263-019-01247-4","volume":"128","author":"L Liu","year":"2020","unstructured":"Liu L, Ouyang W, Wang X, Fieguth P, Chen J, Liu X, Pietik\u00e4inen M (2020) Deep learning for generic object detection: a survey. Int J Comput Vis. 128:261\u2013318","journal-title":"Int J Comput Vis."},{"key":"9430_CR2","doi-asserted-by":"crossref","unstructured":"Bodla N, Singh B, Chellappa R, Davis LS (2017) Soft-NMS\u2013improving object detection with one line of code. In: Proceedings of the IEEE international conference on computer vision, pp 5561\u20135569","DOI":"10.1109\/ICCV.2017.593"},{"key":"9430_CR3","doi-asserted-by":"crossref","unstructured":"Cai Z, Vasconcelos N (2018) Cascade r-cnn: Delving into high quality object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6154\u20136162","DOI":"10.1109\/CVPR.2018.00644"},{"key":"9430_CR4","doi-asserted-by":"crossref","unstructured":"Fan Q, Zhuo W, Tang CK, Tai YW (2020) Few-shot object detection with attention-rpn and multi-relation detector. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4013\u20134022","DOI":"10.1109\/CVPR42600.2020.00407"},{"key":"9430_CR5","doi-asserted-by":"crossref","unstructured":"Hu H, Gu J, Zhang Z, Dai J, Wei Y (2018) Relation networks for object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3588\u20133597","DOI":"10.1109\/CVPR.2018.00378"},{"key":"9430_CR6","doi-asserted-by":"crossref","unstructured":"Lyu P, Liao M, Yao C, Wu W, Bai X (2018) Mask textspotter: An end-to-end trainable neural network for spotting text with arbitrary shapes. In: Proceedings of the European conference on computer vision (ECCV), pp 67\u201383","DOI":"10.1007\/978-3-030-01264-9_5"},{"key":"9430_CR7","doi-asserted-by":"crossref","unstructured":"Girshick R (2015) Fast R-CNN. In: Proceedings of the IEEE international conference on computer vision, pp 1440\u20131448","DOI":"10.1109\/ICCV.2015.169"},{"key":"9430_CR8","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster R-CNN: towards real-time object detection with region proposal networks. Adv Neural Inf Process Syst 28"},{"key":"9430_CR9","doi-asserted-by":"crossref","unstructured":"Liu W, Anguelov D, Erhan D, Szegedy C, Reed S, Fu CY, Berg AC (2016) Ssd: single shot multibox detector. In: Computer vision\u2013ECCV 2016: 14th European conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14, Springer, pp 21\u201337","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"9430_CR10","doi-asserted-by":"crossref","unstructured":"Redmon J, Farhadi A (2017) Yolo9000: better, faster, stronger. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7263\u20137271","DOI":"10.1109\/CVPR.2017.690"},{"key":"9430_CR11","unstructured":"Redmon J, Farhadi A (2018) Yolov3: an incremental improvement. arXiv preprint arXiv:1804.02767"},{"key":"9430_CR12","unstructured":"Bochkovskiy A, Wang CY, Liao HYM (2020) Yolov4: optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934"},{"key":"9430_CR13","unstructured":"Li C, Li L, Jiang H, Weng K, Geng Y, Li L, Ke Z, Li Q, Cheng M, Nie W, et\u00a0al. (2022) Yolov6: a single-stage object detection framework for industrial applications. arXiv preprint arXiv:2209.02976"},{"key":"9430_CR14","doi-asserted-by":"crossref","unstructured":"Wang CY, Bochkovskiy A, Liao HYM (2022) Yolov7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. arXiv preprint arXiv:2207.02696","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"9430_CR15","unstructured":"Ge Z, Liu S, Wang F, Li Z, Sun J (2021) Yolox: Exceeding yolo series in 2021. arXiv preprint arXiv:2107.08430"},{"key":"9430_CR16","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"9430_CR17","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: Computer vision\u2013ECCV 2020: 16th European conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part I 16, Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"9430_CR18","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"9430_CR19","unstructured":"Zhu X, Su W, Lu L, Li B, Wang X, Dai J (2020) Deformable DETR: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159"},{"key":"9430_CR20","doi-asserted-by":"crossref","unstructured":"Meng D, Chen X, Fan Z, Zeng G, Li H, Yuan Y, Sun L, Wang J (2021) Conditional DETR for fast training convergence. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3651\u20133660","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"9430_CR21","doi-asserted-by":"crossref","unstructured":"Gao P, Zheng M, Wang X, Dai J, Li H (2021) Fast convergence of DETR with spatially modulated co-attention. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3621\u20133630","DOI":"10.1109\/ICCV48922.2021.00360"},{"key":"9430_CR22","unstructured":"Zhang H, Li F, Liu S, Zhang L, Su H, Zhu J, Ni LM, Shum HY (2022a) Dino: DETR with improved DeNoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605"},{"key":"9430_CR23","doi-asserted-by":"crossref","unstructured":"Zhang G, Luo Z, Yu Y, Cui K, Lu S (2022b) Accelerating DETR convergence via semantic-aligned matching. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 949\u2013958","DOI":"10.1109\/CVPR52688.2022.00102"},{"key":"9430_CR24","unstructured":"Jain V, Learned-Miller E (2010) Fddb: A benchmark for face detection in unconstrained settings. Tech. rep, UMass Amherst technical report"},{"key":"9430_CR25","doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, Springer, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"issue":"7","key":"9430_CR26","doi-asserted-by":"publisher","first-page":"3400","DOI":"10.3390\/s23073400","volume":"23","author":"G Hamano","year":"2023","unstructured":"Hamano G, Imaizumi S, Kiya H (2023) Effects of jpeg compression on vision transformer image classification for encryption-then-compression images. Sensors 23(7):3400","journal-title":"Sensors"},{"key":"9430_CR27","doi-asserted-by":"crossref","unstructured":"Roy SK, Deria A, Hong D, Rasti B, Plaza A, Chanussot J (2023) Multimodal fusion transformer for remote sensing image classification. IEEE Trans Geosci Remote Sens","DOI":"10.1109\/TGRS.2023.3286826"},{"issue":"11","key":"9430_CR28","doi-asserted-by":"publisher","first-page":"3003","DOI":"10.1109\/TMI.2022.3176598","volume":"41","author":"Y Zheng","year":"2022","unstructured":"Zheng Y, Gindra RH, Green EJ, Burks EJ, Betke M, Beane JE, Kolachalama VB (2022) A graph-transformer for whole slide image classification. IEEE Trans Med Imaging 41(11):3003\u20133015","journal-title":"IEEE Trans Med Imaging"},{"key":"9430_CR29","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han K, Xiao A, Wu E, Guo J, Xu C, Wang Y (2021) Transformer in transformer. Adv Neural Inf Process Syst 34:15908\u201315919","journal-title":"Adv Neural Inf Process Syst"},{"key":"9430_CR30","first-page":"2938","volume":"36","author":"X Xu","year":"2022","unstructured":"Xu X, Xu N (2022) Hierarchical image generation via transformer-based sequential patch selection. Proc AAAI Conf Artif Intell 36:2938\u20132945","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"9430_CR31","doi-asserted-by":"crossref","unstructured":"Esser P, Rombach R, Ommer B (2021) Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12873\u201312883","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"9430_CR32","doi-asserted-by":"crossref","unstructured":"Zhang B, Gu S, Zhang B, Bao J, Chen D, Wen F, Wang Y, Guo B (2022) Styleswin: Transformer-based GAN for high-resolution image generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11304\u201311314","DOI":"10.1109\/CVPR52688.2022.01102"},{"key":"9430_CR33","doi-asserted-by":"crossref","unstructured":"Chang H, Zhang H, Jiang L, Liu C, Freeman WT (2022) Maskgit: Masked generative image transformer. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11315\u201311325","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"9430_CR34","doi-asserted-by":"crossref","unstructured":"Plizzari C, Cannici M, Matteucci M (2021a) Spatial temporal transformer network for skeleton-based action recognition. In: Pattern recognition. ICPR international workshops and challenges: virtual event, January 10\u201315, 2021, Proceedings, Part III, Springer, pp 694\u2013701","DOI":"10.1007\/978-3-030-68796-0_50"},{"key":"9430_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2021.103219","volume":"208","author":"C Plizzari","year":"2021","unstructured":"Plizzari C, Cannici M, Matteucci M (2021) Skeleton-based action recognition via spatial and temporal transformer networks. Comput Vis Image Underst 208:103219","journal-title":"Comput Vis Image Underst"},{"issue":"1","key":"9430_CR36","doi-asserted-by":"publisher","first-page":"246","DOI":"10.1109\/TCDS.2020.3048883","volume":"14","author":"X Li","year":"2021","unstructured":"Li X, Hou Y, Wang P, Gao Z, Xu M, Li W (2021) Trear: transformer-based RGB-D egocentric action recognition. IEEE Trans Cognit Dev Syst 14(1):246\u2013252","journal-title":"IEEE Trans Cognit Dev Syst"},{"key":"9430_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.measurement.2022.111228","volume":"196","author":"S Yu","year":"2022","unstructured":"Yu S, Wang M, Pang S, Song L, Qiao S (2022) Intelligent fault diagnosis and visual interpretability of rotating machinery based on residual neural network. Measurement 196:111228","journal-title":"Measurement"},{"key":"9430_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.ymssp.2022.109789","volume":"185","author":"S Yu","year":"2023","unstructured":"Yu S, Wang M, Pang S, Song L, Zhai X, Zhao Y (2023) Tdmsae: a transferable decoupling multi-scale autoencoder for mechanical fault diagnosis. Mech Syst Signal Process 185:109789","journal-title":"Mech Syst Signal Process"},{"key":"9430_CR39","unstructured":"Zhao G, Lin J, Zhang Z, Ren X, Sun X (2019) Sparse transformer: concentrated attention through explicit selection"},{"key":"9430_CR40","unstructured":"Wang S, Li BZ, Khabsa M, Fang H, Ma H (2020) Linformer: self-attention with linear complexity. arXiv preprint arXiv:2006.04768"},{"key":"9430_CR41","doi-asserted-by":"crossref","unstructured":"Messina N, Falchi F, Esuli A, Amato G (2021) Transformer reasoning network for image-text matching and retrieval. In: 2020 25th international conference on pattern recognition (ICPR), IEEE, pp 5222\u20135229","DOI":"10.1109\/ICPR48806.2021.9413172"},{"key":"9430_CR42","doi-asserted-by":"crossref","unstructured":"Mueller J, Thyagarajan A (2016) Siamese recurrent architectures for learning sentence similarity. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a030","DOI":"10.1609\/aaai.v30i1.10350"},{"key":"9430_CR43","doi-asserted-by":"crossref","unstructured":"Chen Q, Zhu X, Ling Z, Wei S, Jiang H, Inkpen D (2016) Enhanced LSTM for natural language inference. arXiv preprint arXiv:1609.06038","DOI":"10.18653\/v1\/P17-1152"},{"key":"9430_CR44","doi-asserted-by":"crossref","unstructured":"Chen H, Luo Z, Zhou L, Tian Y, Zhen M, Fang T, Mckinnon D, Tsin Y, Quan L (2022) Aspanformer: detector-free image matching with adaptive span transformer. In: European conference on computer vision, Springer, pp 20\u201336","DOI":"10.1007\/978-3-031-19824-3_2"},{"key":"9430_CR45","doi-asserted-by":"publisher","first-page":"445","DOI":"10.1016\/j.inffus.2022.10.030","volume":"91","author":"J Chen","year":"2023","unstructured":"Chen J, Chen X, Chen S, Liu Y, Rao Y, Yang Y, Wang H, Wu D (2023) Shape-former: Bridging CNN and transformer via ShapeConv for multimodal image matching. Inf. Fusion 91:445\u2013457","journal-title":"Inf. Fusion"},{"key":"9430_CR46","first-page":"1992","volume":"34","author":"S Liao","year":"2021","unstructured":"Liao S, Shao L (2021) Transmatcher: deep image matching through transformers for generalizable person re-identification. Adv Neural Inf Process Syst 34:1992\u20132003","journal-title":"Adv Neural Inf Process Syst"},{"key":"9430_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109443","volume":"139","author":"W Su","year":"2023","unstructured":"Su W, Wang Y, Li K, Gao P, Qiao Y (2023) Hybrid token transformer for deep face recognition. Pattern Recogn 139:109443","journal-title":"Pattern Recogn"},{"issue":"8","key":"9430_CR48","doi-asserted-by":"publisher","first-page":"1126","DOI":"10.3390\/agriculture12081126","volume":"12","author":"X Li","year":"2022","unstructured":"Li X, Du J, Yang J, Li S (2022) When mobilenetv2 meets transformer: a balanced sheep face recognition model. Agriculture 12(8):1126","journal-title":"Agriculture"},{"key":"9430_CR49","doi-asserted-by":"publisher","first-page":"2095","DOI":"10.1109\/TIFS.2022.3177960","volume":"17","author":"M Luo","year":"2022","unstructured":"Luo M, Wu H, Huang H, He W, He R (2022) Memory-modulated transformer network for heterogeneous face recognition. IEEE Trans Inf Forensics Secur 17:2095\u20132109","journal-title":"IEEE Trans Inf Forensics Secur"},{"key":"9430_CR50","doi-asserted-by":"crossref","unstructured":"Chopra S, Hadsell R, LeCun Y (2005) Learning a similarity metric discriminatively, with application to face verification. In: 2005 IEEE computer society conference on computer vision and pattern recognition (CVPR\u201905), IEEE, vol\u00a01, pp 539\u2013546","DOI":"10.1109\/CVPR.2005.202"},{"key":"9430_CR51","unstructured":"Koch G, Zemel R, Salakhutdinov R, et\u00a0al. (2015) Siamese neural networks for one-shot image recognition. In: ICML deep learning workshop, Lille, vol\u00a02"},{"key":"9430_CR52","doi-asserted-by":"crossref","unstructured":"Lin TY, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2117\u20132125","DOI":"10.1109\/CVPR.2017.106"},{"key":"9430_CR53","doi-asserted-by":"crossref","unstructured":"Liu S, Qi L, Qin H, Shi J, Jia J (2018) Path aggregation network for instance segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 8759\u20138768","DOI":"10.1109\/CVPR.2018.00913"},{"key":"9430_CR54","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der\u00a0Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"9430_CR55","doi-asserted-by":"crossref","unstructured":"Zhou B, Khosla A, Lapedriza A, Oliva A, Torralba A (2016) Learning deep features for discriminative localization. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2921\u20132929","DOI":"10.1109\/CVPR.2016.319"},{"key":"9430_CR56","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask R-CNN. In: Proceedings of the IEEE international conference on computer vision, pp 2961\u20132969","DOI":"10.1109\/ICCV.2017.322"},{"key":"9430_CR57","unstructured":"Loshchilov I, Hutter F (2017) Fixing weight decay regularization in Adam"},{"key":"9430_CR58","doi-asserted-by":"crossref","unstructured":"Hu J, Shen L, Sun G (2018) Squeeze-and-excitation networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7132\u20137141","DOI":"10.1109\/CVPR.2018.00745"},{"key":"9430_CR59","doi-asserted-by":"crossref","unstructured":"Woo S, Park J, Lee JY, Kweon IS (2018) Cbam: convolutional block attention module. In: Proceedings of the European conference on computer vision (ECCV), pp 3\u201319","DOI":"10.1007\/978-3-030-01234-2_1"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09430-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-09430-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09430-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,16]],"date-time":"2024-03-16T19:06:26Z","timestamp":1710615986000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-09430-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,9]]},"references-count":59,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,4]]}},"alternative-id":["9430"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-09430-6","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,9]]},"assertion":[{"value":"6 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 February 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}