{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T16:02:59Z","timestamp":1774540979359,"version":"3.50.1"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2023,12,23]],"date-time":"2023-12-23T00:00:00Z","timestamp":1703289600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,23]],"date-time":"2023-12-23T00:00:00Z","timestamp":1703289600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1007\/s11760-023-02899-z","type":"journal-article","created":{"date-parts":[[2023,12,23]],"date-time":"2023-12-23T15:02:03Z","timestamp":1703343723000},"page":"2265-2275","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Ctnet: rethinking convolutional neural networks and vision transformer for medical image segmentation"],"prefix":"10.1007","volume":"18","author":[{"given":"Zhixin","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuhao","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuhua","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,23]]},"reference":[{"key":"2899_CR1","doi-asserted-by":"crossref","unstructured":"Ates, G.C., Mohan, P., Celik, E.: Dual cross-attention for medical image segmentation. arXiv preprint arXiv:2303.17696 (2023)","DOI":"10.1016\/j.engappai.2023.107139"},{"key":"2899_CR2","doi-asserted-by":"crossref","unstructured":"Cao, H., Wang, Y., Chen, J. et\u00a0al.: Swin-unet: Unet-like pure transformer for medical image segmentation. In: European Conference on Computer Vision, Springer, pp 205\u2013218 (2022)","DOI":"10.1007\/978-3-031-25066-8_9"},{"key":"2899_CR3","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G. et\u00a0al.: End-to-end object detection with transformers. In: European Conference on Computer Vision, Springer, pp 213\u2013229 (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2899_CR4","doi-asserted-by":"crossref","unstructured":"Chen, H., Wang, Y., Guo, T. et\u00a0al.: Pre-trained image processing transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 12,299\u201312,310 (2021a)","DOI":"10.1109\/CVPR46437.2021.01212"},{"key":"2899_CR5","unstructured":"Chen, J., Lu, Y., Yu, Q. et\u00a0al.: Transunet: transformers make strong encoders for medical image segmentation. arXiv preprint arXiv:2102.04306 (2021b)"},{"key":"2899_CR6","unstructured":"Chu, X., Tian, Z., Zhang, B. et\u00a0al.: Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882 (2021)"},{"key":"2899_CR7","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R. et\u00a0al.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, IEEE, pp 248\u2013255 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2899_CR8","unstructured":"Devlin, J., Chang, M.W., Lee, K. et\u00a0al.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"2899_CR9","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A. et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"2899_CR10","unstructured":"El-Nouby, A., Neverova, N., Laptev, I. et\u00a0al.: Training vision transformers for image retrieval. arXiv preprint arXiv:2102.05644 (2021)"},{"issue":"1","key":"2899_CR11","doi-asserted-by":"publisher","first-page":"407","DOI":"10.1016\/j.cmpb.2012.03.009","volume":"108","author":"MM Fraz","year":"2012","unstructured":"Fraz, M.M., Remagnino, P., Hoppe, A., et al.: Blood vessel segmentation methodologies in retinal images-a survey. Comput. Methods Programs Biomed. 108(1), 407\u2013433 (2012)","journal-title":"Comput. Methods Programs Biomed."},{"issue":"12","key":"2899_CR12","doi-asserted-by":"publisher","first-page":"5990","DOI":"10.3390\/app12125990","volume":"12","author":"Y Gulzar","year":"2022","unstructured":"Gulzar, Y., Khan, S.A.: Skin lesion segmentation based on vision transformers and convolutional neural networks-a comparative study. Appl. Sci. 12(12), 5990 (2022)","journal-title":"Appl. Sci."},{"key":"2899_CR13","doi-asserted-by":"crossref","unstructured":"Han, K., Wang, Y., Tian, Q. et\u00a0al.: Ghostnet: more features from cheap operations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 1580\u20131589 (2020)","DOI":"10.1109\/CVPR42600.2020.00165"},{"key":"2899_CR14","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han, K., Xiao, A., Wu, E., et al.: Transformer in transformer. Adv. Neural. Inf. Process. Syst. 34, 15908\u201315919 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2899_CR15","doi-asserted-by":"crossref","unstructured":"Hatamizadeh, A., Tang, Y., Nath, V. et\u00a0al.: Unetr: transformers for 3d medical image segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 574\u2013584 (2022)","DOI":"10.1109\/WACV51458.2022.00181"},{"key":"2899_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S. et\u00a0al.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2899_CR17","unstructured":"Hu, J., Shen, L., Albanie, S. et\u00a0al.: Gather-excite: exploiting feature context in convolutional neural networks. Adv. Neural Inf. Process. Syst. 31 (2018a)"},{"key":"2899_CR18","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7132\u20137141 (2018b)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"2899_CR19","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der\u00a0Maaten, L. et\u00a0al.: Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4700\u20134708 (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"2899_CR20","unstructured":"Hung, P.V., Duy\u00a0Manh, N., Oanh, N.T. et\u00a0al.: Ugcanet: a unified global context-aware transformer-based network with feature alignment for endoscopic image analysis. arXiv e-prints pp arXiv\u20132307 (2023)"},{"key":"2899_CR21","unstructured":"Iandola, F.N., Han, S., Moskewicz, M.W. et\u00a0al.: Squeezenet: alexnet-level accuracy with 50x fewer parameters and 0.5 mb model size. arXiv preprint arXiv:1602.07360 (2016)"},{"key":"2899_CR22","doi-asserted-by":"crossref","unstructured":"Jha, D., Smedsrud, P.H., Riegler, M.A. et\u00a0al.: Kvasir-seg: a segmented polyp dataset. In: International Conference on Multimedia Modeling, Springer, pp 451\u2013462 (2020)","DOI":"10.1007\/978-3-030-37734-2_37"},{"key":"2899_CR23","doi-asserted-by":"crossref","unstructured":"Krause, J., Stark, M., Deng, J. et\u00a0al.: 3d object representations for fine-grained categorization. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp 554\u2013561 (2013)","DOI":"10.1109\/ICCVW.2013.77"},{"key":"2899_CR24","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural Inf. Process. Syst. 25 (2012)"},{"issue":"5","key":"2899_CR25","doi-asserted-by":"publisher","first-page":"1380","DOI":"10.1109\/TMI.2019.2947628","volume":"39","author":"N Kumar","year":"2019","unstructured":"Kumar, N., Verma, R., Anand, D., et al.: A multi-organ nucleus segmentation challenge. IEEE Trans. Med. Imaging 39(5), 1380\u20131391 (2019)","journal-title":"IEEE Trans. Med. Imaging"},{"issue":"11","key":"2899_CR26","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., et al.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"2899_CR27","doi-asserted-by":"crossref","unstructured":"Li, Z., Li, Y., Li, Q. et\u00a0al.: Lvit: language meets vision transformer in medical image segmentation. IEEE Transact. Med. Imaging (2023)","DOI":"10.1109\/TMI.2023.3291719"},{"key":"2899_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y. et\u00a0al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10,012\u201310,022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2899_CR29","unstructured":"Ramachandran, P., Parmar, N., Vaswani, A. et\u00a0al.: Stand-alone self-attention in vision models. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"2899_CR30","unstructured":"Reed, S., Lee, H., Anguelov, D. et\u00a0al.: Training deep neural networks on noisy labels with bootstrapping. arXiv preprint arXiv:1412.6596 (2014)"},{"key":"2899_CR31","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M. et\u00a0al.: Mobilenetv2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"2899_CR32","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1016\/j.media.2019.01.012","volume":"53","author":"J Schlemper","year":"2019","unstructured":"Schlemper, J., Oktay, O., Schaap, M., et al.: Attention gated networks: learning to leverage salient regions in medical images. Med. Image Anal. 53, 197\u2013207 (2019)","journal-title":"Med. Image Anal."},{"key":"2899_CR33","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"2899_CR34","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S. et\u00a0al.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 2818\u20132826 (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"2899_CR35","unstructured":"Tan, M., Le, Q.: Efficientnet: rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, PMLR, pp 6105\u20136114 (2019)"},{"key":"2899_CR36","unstructured":"Touvron, H., Cord, M., Douze, M. et\u00a0al.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, PMLR, pp 10,347\u201310,357 (2021)"},{"key":"2899_CR37","doi-asserted-by":"crossref","unstructured":"Tragakis, A., Kaul, C., Murray-Smith, R. et\u00a0al.: The fully convolutional transformer for medical image segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 3660\u20133669 (2023)","DOI":"10.1109\/WACV56688.2023.00365"},{"key":"2899_CR38","doi-asserted-by":"crossref","unstructured":"Valanarasu, J.M.J., Oza, P., Hacihaliloglu, I. et\u00a0al.: Medical transformer: gated axial-attention for medical image segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, Springer, pp 36\u201346 (2021)","DOI":"10.1007\/978-3-030-87193-2_4"},{"key":"2899_CR39","unstructured":"Vaswani, A., Shazeer, N., Parmar, N. et\u00a0al.: Attention is all you need. Adv Neural Inf Process Syst 30 (2017)"},{"key":"2899_CR40","doi-asserted-by":"crossref","unstructured":"Wang, F., Jiang, M., Qian, C. et\u00a0al.: Residual attention network for image classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 3156\u20133164 (2017)","DOI":"10.1109\/CVPR.2017.683"},{"key":"2899_CR41","doi-asserted-by":"crossref","unstructured":"Wang, W., Chen, C., Ding, M. et\u00a0al.: Transbts: multimodal brain tumor segmentation using transformer. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, Springer, pp 109\u2013119 (2021a)","DOI":"10.1007\/978-3-030-87193-2_11"},{"key":"2899_CR42","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X. et\u00a0al.: Pyramid vision transformer: a versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 568\u2013578 (2021b)","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"2899_CR43","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A. et\u00a0al.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 7794\u20137803 (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"2899_CR44","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.Y. et\u00a0al.: Cbam: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"2899_CR45","doi-asserted-by":"crossref","unstructured":"Wu, H., Xiao, B., Codella, N. et\u00a0al.: Cvt: Introducing convolutions to vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 22\u201331 (2021)","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"2899_CR46","doi-asserted-by":"crossref","unstructured":"Xie, Y., Zhang, J., Shen, C. et\u00a0al.: Cotr: efficiently bridging cnn and transformer for 3d medical image segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, Springer, pp 171\u2013180 (2021)","DOI":"10.1007\/978-3-030-87199-4_16"},{"issue":"109","key":"2899_CR47","first-page":"228","volume":"136","author":"F Yuan","year":"2023","unstructured":"Yuan, F., Zhang, Z., Fang, Z.: An effective cnn and transformer complementary network for medical image segmentation. Pattern Recogn. 136(109), 228 (2023)","journal-title":"Pattern Recogn."},{"key":"2899_CR48","doi-asserted-by":"crossref","unstructured":"Yuan, K., Guo, S., Liu, Z. et\u00a0al.: Incorporating convolution designs into visual transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 579\u2013588 (2021a)","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"2899_CR49","doi-asserted-by":"crossref","unstructured":"Yuan, L., Chen, Y., Wang, T. et\u00a0al.: Tokens-to-token vit: training vision transformers from scratch on imagenet. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 558\u2013567 (2021b)","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"2899_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Liu, H., Hu, Q.: Transfuse: fusing transformers and cnns for medical image segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, Springer, pp 14\u201324 (2021)","DOI":"10.1007\/978-3-030-87193-2_2"},{"key":"2899_CR51","doi-asserted-by":"crossref","unstructured":"Zheng, S., Lu, J., Zhao, H.: et\u00a0al Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 6881\u20136890 (2021)","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"2899_CR52","unstructured":"Zhu, X., Su, W., Lu, L. et\u00a0al.: Deformable detr: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02899-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-023-02899-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02899-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,19]],"date-time":"2024-03-19T20:17:13Z","timestamp":1710879433000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-023-02899-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,23]]},"references-count":52,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,4]]}},"alternative-id":["2899"],"URL":"https:\/\/doi.org\/10.1007\/s11760-023-02899-z","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12,23]]},"assertion":[{"value":"12 January 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 November 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 November 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 December 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}