{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T21:35:24Z","timestamp":1780608924720,"version":"3.54.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"25","license":[{"start":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T00:00:00Z","timestamp":1706054400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T00:00:00Z","timestamp":1706054400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Scientific and Technological Innovation Plan of Shanghai STC","award":["21511102605"],"award-info":[{"award-number":["21511102605"]}]},{"name":"Wuxi Municipal Health Commission Translational Medicine Research Project","award":["ZH202102"],"award-info":[{"award-number":["ZH202102"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-18234-8","type":"journal-article","created":{"date-parts":[[2024,1,24]],"date-time":"2024-01-24T05:03:11Z","timestamp":1706072591000},"page":"67213-67229","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Dynamic multi-headed self-attention and multiscale enhancement vision transformer for object detection"],"prefix":"10.1007","volume":"83","author":[{"given":"Sikai","family":"Fang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaofeng","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifan","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangling","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuefeng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,1,24]]},"reference":[{"key":"18234_CR1","doi-asserted-by":"crossref","unstructured":"Redmon J, Divvala S, Girshick R, Farhadi A (2016) You only look once: Unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 779\u2013788","DOI":"10.1109\/CVPR.2016.91"},{"issue":"10","key":"18234_CR2","doi-asserted-by":"publisher","first-page":"1345","DOI":"10.1109\/TKDE.2009.191","volume":"22","author":"S-J Pan","year":"2010","unstructured":"Pan S-J, Yang Q (2010) A survey on transfer learning. IEEE Trans Knowl Data Eng 22(10):1345\u20131359","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"18234_CR3","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P, Tu Z, He K (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 5987\u20135995","DOI":"10.1109\/CVPR.2017.634"},{"key":"18234_CR4","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger K-Q (2017) Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2261\u20132269","DOI":"10.1109\/CVPR.2017.243"},{"key":"18234_CR5","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Doll\u00e1r P, Girshick R, He K, Hariharan B, Belongie S (2017) Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 936\u2013944","DOI":"10.1109\/CVPR.2017.106"},{"key":"18234_CR6","unstructured":"Luo W, Li Y, Urtasun R, Zemel R (2016) Understanding the effective receptive field in deep convolutional neural networks. Advances in Neural Information Processing Systems, pp 4898\u20134906"},{"key":"18234_CR7","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez A-N, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Advances in Neural Information Processing Systems, pp 5998\u20136008"},{"key":"18234_CR8","doi-asserted-by":"crossref","unstructured":"Srinivas, Lin T-Y, Parmar N, Shlens J, Abbeel P, Vaswani A (2021) Bottleneck transformers for visual recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 16514\u201316524","DOI":"10.1109\/CVPR46437.2021.01625"},{"key":"18234_CR9","doi-asserted-by":"crossref","unstructured":"Pan X, Ge C, Lu R, Song S, Chen G, Huang Z, Huang G (2022) On the integration of self-attention and convolution. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 805\u2013815","DOI":"10.1109\/CVPR52688.2022.00089"},{"key":"18234_CR10","doi-asserted-by":"publisher","first-page":"461","DOI":"10.1109\/TIP.2019.2919937","volume":"29","author":"S Zhou","year":"2019","unstructured":"Zhou S, Nie D, Adeli E, Yin J, Lian J, Shen D (2019) High-resolution encoder\u2013decoder networks for low-contrast medical image segmentation. IEEE Trans Image Process 29:461\u2013475","journal-title":"IEEE Trans Image Process"},{"issue":"12","key":"18234_CR11","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan V, Kendall A, Cipolla R (2017) Segnet: A deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans Pattern Anal Mach Intell 39(12):2481\u20132495","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18234_CR12","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, Uszkoreit J, Houlsby N (2021) An image is worth 16x16 words: Transformers for image recognition at scale. In: Proceedings of the International Conference on Learning Representations, pp 1\u201321"},{"issue":"3","key":"18234_CR13","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky O, Deng J, Su H, Krause J, Satheesh S, Ma S, Huang Z, Karpathy A, Khosla A, Bernstein M, Berg AC, Fei-Fei L (2015) Imagenet large scale visual recognition challenge. Int J Comput Vis 115(3):211\u2013252","journal-title":"Int J Comput Vis"},{"key":"18234_CR14","doi-asserted-by":"crossref","unstructured":"Huang Z, Wang X, Huang L, Huang C, Wei Y, Liu W (2019) Ccnet: Criss-cross attention for semantic segmentation. In: Proceedings of the IEEE International Conference on Computer Vision, pp 603\u2013612","DOI":"10.1109\/ICCV.2019.00069"},{"key":"18234_CR15","doi-asserted-by":"crossref","unstructured":"Fan H, Xiong B, Mangalam K, Li Y, Yan Z, Malik J, Feichtenhofer C (2021) Multiscale vision transformers. In: Proceedings of the IEEE International Conference on Computer Vision, pp 6804\u20136815","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"18234_CR16","doi-asserted-by":"crossref","unstructured":"Tu Z, Talebi H, Zhang H, Yang F, Milanfar P, Bovik A, Li Y (2022) Maxvit: Multi-axis vision transformer. In: Proceedings of the European Conference on Computer Vision, pp 459\u2013479","DOI":"10.1007\/978-3-031-20053-3_27"},{"key":"18234_CR17","doi-asserted-by":"crossref","unstructured":"Chen Z, Li Y, Bengio S, Si S (2019) You look twice: Gaternet for dynamic filter selection in cnns. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 9164\u20139172","DOI":"10.1109\/CVPR.2019.00939"},{"key":"18234_CR18","doi-asserted-by":"crossref","unstructured":"Wang X, Yu F, Dou Z-Y, Darrell T, Gonzalez J-E (2018) Skipnet: Learning dynamic routing in convolutional networks. In: Proceedings of the European Conference on Computer Vision, pp 409\u2013424","DOI":"10.1007\/978-3-030-01261-8_25"},{"key":"18234_CR19","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: Proceedings of the European Conference on Computer Vision, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"issue":"2","key":"18234_CR20","doi-asserted-by":"publisher","first-page":"386","DOI":"10.1109\/TPAMI.2018.2844175","volume":"42","author":"K He","year":"2020","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2020) Mask r-cnn. IEEE Trans Pattern Anal Mach Intell 42(2):386\u2013397","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"18234_CR21","doi-asserted-by":"crossref","unstructured":"Cai Z, Vasconcelos N (2018) Cascade r-cnn: Delving into high quality object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6154\u20136162","DOI":"10.1109\/CVPR.2018.00644"},{"key":"18234_CR22","doi-asserted-by":"crossref","unstructured":"Wang W, Xie E, Li X, Fan D-P, Song K, Liang D, Lu T, Luo P, Shao L (2021) Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE International Conference on Computer Vision, pp 548\u2013558","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"18234_CR23","doi-asserted-by":"crossref","unstructured":"Chen Y, Dai X, Chen D, Liu M, Dong X, Yuan L, Liu Z (2022) Mobile-former: Bridging mobilenet and transformer. In: Proceedings of the IEEE International Conference on Computer Vision, pp 5270\u20135279","DOI":"10.1109\/CVPR52688.2022.00520"},{"key":"18234_CR24","unstructured":"Howard A-G, Zhu M, Chen B, Kalenichenko D, Wang W, Weyand T, Andreetto M, Adam H (2017) Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv: 1704.04861"},{"key":"18234_CR25","unstructured":"Han K, Xiao A, Wu E, Guo J, Xu C, Wang Y (2021) Transformer in transformer. Advances in Neural Information Processing Systems, pp 15908\u201315919"},{"key":"18234_CR26","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE International Conference on Computer Vision, pp 9992\u201310002","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"18234_CR27","doi-asserted-by":"crossref","unstructured":"Zong Z, Song G, Liu Y. Detrs with collaborative hybrid assignments training[C] Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2023: 6748\u20136758.","DOI":"10.1109\/ICCV51070.2023.00621"},{"key":"18234_CR28","unstructured":"Vasu P K A, Gabriel J, Zhu J, Tuzel O, Ranjan A (2023) FastViT: A fast hybrid vision transformer using structural reparameterization. In: Proceedings of the IEEE International Conference on Computer Vision, pp 5785\u20135795"},{"key":"18234_CR29","unstructured":"Hassibi B, Stork D (1993) Second order derivatives for network pruning: Optimal brain surgeon. Advances in Neural Information Processing Systems, pp 164\u2013171"},{"key":"18234_CR30","doi-asserted-by":"crossref","unstructured":"Wu J, Leng C, Wang Y, Hu Q, Cheng J (2016) Quantized convolutional neural networks for mobile devices. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4820\u20134828","DOI":"10.1109\/CVPR.2016.521"},{"key":"18234_CR31","unstructured":"Han S, Mao H, Dally W-J (2016) Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. In: Proceedings of the International Conference on Learning Representations, pp 1\u201314"},{"key":"18234_CR32","doi-asserted-by":"crossref","unstructured":"Li Y, Song L, Chen Y, Li Z, Zhang X, Wang X, Sun J (2020) Learning dynamic routing for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8550\u20138559","DOI":"10.1109\/CVPR42600.2020.00858"},{"key":"18234_CR33","unstructured":"Bolukbasi T, Wang J, Dekel O, Saligrama V (2017) Adaptive neural networks for efficient inference. In: Proceedings of the International Conference on Machine Learning, pp 527\u2013536"},{"issue":"120","key":"18234_CR34","first-page":"1","volume":"23","author":"W Fedus","year":"2022","unstructured":"Fedus W, Zoph B, Shazeer N (2022) Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity. J Mach Learn Res 23(120):1\u201339","journal-title":"J Mach Learn Res"},{"key":"18234_CR35","doi-asserted-by":"crossref","unstructured":"Lin Z, Wang Y, Zhang J et al (2023) DynamicDet: A Unified Dynamic Architecture for Object Detection[C] Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6282\u20136291.","DOI":"10.1109\/CVPR52729.2023.00608"},{"key":"18234_CR36","unstructured":"Rao Y, Zhao W, Liu B, Lu J, Zhou J, Hsieh C-J (2021) Dynamicvit: Efficient vision transformers with dynamic token sparsification. Advances in Neural Information Processing Systems, pp 13937\u201313949"},{"key":"18234_CR37","doi-asserted-by":"crossref","unstructured":"Selvaraju R-R, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-cam: Visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp 618\u2013626","DOI":"10.1109\/ICCV.2017.74"},{"key":"18234_CR38","unstructured":"Krizhevsky A, Hinton G (2009) Learning multiple layers of features from tiny images. Personal Communication, 2009"},{"key":"18234_CR39","unstructured":"Zhu X, Su W, Lu L, Li B, Wang X, Dai J (2021) Deformable detr: Deformable transformers for end-to-end object detection. In: Proceedings of the International Conference on Learning Representations, pp 1\u201316"},{"key":"18234_CR40","doi-asserted-by":"crossref","unstructured":"Gu J, Kwon H, Wang D, Ye W, Li M, Chen Y-H, Lai L, Chandra V, Pan D-Z (2022) Multi-scale high-resolution vision transformer for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 12084\u201312093","DOI":"10.1109\/CVPR52688.2022.01178"},{"key":"18234_CR41","doi-asserted-by":"crossref","unstructured":"Li Y, Mao H, Girshick R, He K (2022) Exploring plain vision transformer backbones for object detection. In: Proceedings of the European Conference on Computer Vision, pp 280\u2013296","DOI":"10.1007\/978-3-031-20077-9_17"},{"key":"18234_CR42","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick C-L (2014) Microsoft coco: Common objects in context. In: Proceedings of the European Conference on Computer Vision, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"18234_CR43","doi-asserted-by":"crossref","unstructured":"Yuan L, Chen Y, Wang T, Yu W, Shi Y, Jiang Z-H., Tay F-E-H, Feng J, Yan S (2021) Tokens-to-token vit: Training vision transformers from scratch on imagenet. In: Proceedings of the IEEE International Conference on Computer Vision, pp 538\u2013547","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"18234_CR44","unstructured":"Chu X, Tian Z, Zhang B, Wang X, Wei X, Xia H, Shen C (2021) Conditional positional encodings for vision transformers. In: Proceedings of the International Conference on Learning Representations, pp 1\u201319"},{"key":"18234_CR45","unstructured":"Chu X, Tian Z, Wang Y, Zhang B, Ren H, Wei X, Xia H, Shen C (2021) Twins: Revisiting the design of spatial attention in vision transformers. Advances in Neural Information Processing Systems, pp 9355\u20139366"},{"key":"18234_CR46","doi-asserted-by":"crossref","unstructured":"Yuan K, Guo S, Liu Z, Zhou A, Yu F, Wu W (2021) Incorporating convolution designs into visual transformers. In: Proceedings of the IEEE International Conference on Computer Vision, pp 559\u2013568","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"18234_CR47","doi-asserted-by":"crossref","unstructured":"Srinivas A, Lin T-Y, Parmar N, Shlens J, Abbeel P, Vaswani A (2021) Bottleneck transformers for visual recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 16514\u201316524","DOI":"10.1109\/CVPR46437.2021.01625"},{"key":"18234_CR48","unstructured":"Tan M, Le Q (2021) Efficientnetv2: Smaller models and faster training. In: Proceedings of the International Conference on Machine Learning, pp 10096\u201310106"},{"key":"18234_CR49","unstructured":"Chi C, Wei F, Hu H (2020) Relationnet++: Bridging visual representations for object detection via transformer decoder. Advances in Neural Information Processing Systems, pp 13564\u201313574"},{"key":"18234_CR50","doi-asserted-by":"crossref","unstructured":"Qiu H, Ma Y, Li Z, Liu S, Sun J (2020) Borderdet: Border feature for dense object detection. In: Proceedings of the European Conference on Computer Vision, pp 549\u2013564","DOI":"10.1007\/978-3-030-58452-8_32"},{"key":"18234_CR51","doi-asserted-by":"crossref","unstructured":"Dai X, Chen Y, Xiao B, Chen D, Liu M, Yuan L, Zhang L (2021) Dynamic head: Unifying object detection heads with attentions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7369\u20137378","DOI":"10.1109\/CVPR46437.2021.00729"},{"key":"18234_CR52","doi-asserted-by":"crossref","unstructured":"Li X, Wang W, Hu X, Li J, Tang J, Yang J (2021) Generalized focal loss v2: Learning reliable localization quality estimation for dense object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 11627\u201311636","DOI":"10.1109\/CVPR46437.2021.01146"},{"key":"18234_CR53","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"18234_CR54","doi-asserted-by":"crossref","unstructured":"Sandler M, Howard A, Zhu M, Zhmoginov A, Chen L-C (2018) Mobilenetv2: Inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4510\u20134520","DOI":"10.1109\/CVPR.2018.00474"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18234-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-18234-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-18234-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,9]],"date-time":"2024-07-09T10:30:05Z","timestamp":1720521005000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-18234-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,24]]},"references-count":54,"journal-issue":{"issue":"25","published-online":{"date-parts":[[2024,7]]}},"alternative-id":["18234"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-18234-8","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,24]]},"assertion":[{"value":"13 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 January 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 January 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interests that are directly or indirectly related to the submission of this publication.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interests"}}]}}