{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T04:39:14Z","timestamp":1784522354694,"version":"3.55.0"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,5,11]],"date-time":"2025-05-11T00:00:00Z","timestamp":1746921600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,11]],"date-time":"2025-05-11T00:00:00Z","timestamp":1746921600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00371-025-03909-6","type":"journal-article","created":{"date-parts":[[2025,5,11]],"date-time":"2025-05-11T08:12:31Z","timestamp":1746951151000},"page":"8975-9003","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["MMTU-Net: enhancing medical image semantic segmentation with multi-level multi-scale fusion and transformer"],"prefix":"10.1007","volume":"41","author":[{"given":"Xilei","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuefei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuquan","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yutong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,11]]},"reference":[{"issue":"1","key":"3909_CR1","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1186\/s12880-022-00793-7","volume":"22","author":"HE Kim","year":"2022","unstructured":"Kim, H.E., Cosa-Linan, A., Santhanam, N., et al.: Transfer learning for medical image classification: a literature review. BMC Med. Imaging 22(1), 69 (2022)","journal-title":"BMC Med. Imaging"},{"issue":"10","key":"3909_CR2","doi-asserted-by":"publisher","first-page":"1822","DOI":"10.1109\/LGRS.2019.2954578","volume":"17","author":"J Han","year":"2019","unstructured":"Han, J., Moradi, S., Faramarzi, I., et al.: A local contrast method for infrared small-target detection utilizing a tri-layer window. IEEE Geosci. Remote Sens. Lett. 17(10), 1822\u20131826 (2019)","journal-title":"IEEE Geosci. Remote Sens. Lett."},{"key":"3909_CR3","unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"3909_CR4","doi-asserted-by":"crossref","unstructured":"Strudel, R., Garcia, R., Laptev, I., et al.: Segmenter: transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7262\u20137272 (2021)","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"3909_CR5","doi-asserted-by":"crossref","unstructured":"Zheng S., Jayasumana S., Romera-Paredes B., et al.: Conditional random fields as recurrent neural networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1529\u20131537 (2015)","DOI":"10.1109\/ICCV.2015.179"},{"key":"3909_CR6","doi-asserted-by":"publisher","first-page":"106548","DOI":"10.1016\/j.ymssp.2019.106548","volume":"138","author":"A Altan","year":"2020","unstructured":"Altan, A., Hac\u0131o\u011flu, R.: Model predictive control of three-axis gimbal system mounted on UAV for real-time target tracking under external disturbances. Mech. Syst. Signal Process. 138, 106548 (2020)","journal-title":"Mech. Syst. Signal Process."},{"key":"3909_CR7","doi-asserted-by":"crossref","unstructured":"Danelljan, M., Robinson A., Shahbaz Khan F., et al.: Beyond correlation filters: learning continuous convolution operators for visual tracking. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part V 14, pp. 472\u2013488. Springer (2016)","DOI":"10.1007\/978-3-319-46454-1_29"},{"key":"3909_CR8","doi-asserted-by":"crossref","unstructured":"Bertinetto, L., Valmadre, J., Henriques, J.F., et al.: Fully-convolutional siamese networks for object tracking. In: Computer Vision\u2013ECCV 2016 Workshops: Amsterdam, The Netherlands, October 8\u201310 and 15\u201316, 2016, Proceedings, Part II 14, pp. 850\u2013865. Springer (2016)","DOI":"10.1007\/978-3-319-48881-3_56"},{"key":"3909_CR9","doi-asserted-by":"publisher","first-page":"5985","DOI":"10.1007\/s11661-020-06008-4","volume":"51","author":"EA Holm","year":"2020","unstructured":"Holm, E.A., Cohn, R., Gao, N., et al.: Overview: computer vision and machine learning for microstructural characterization and analysis. Metall. Mater. Trans. A 51, 5985\u20135999 (2020)","journal-title":"Metall. Mater. Trans. A"},{"issue":"3","key":"3909_CR10","doi-asserted-by":"publisher","first-page":"1341","DOI":"10.1109\/TITS.2020.2972974","volume":"22","author":"D Feng","year":"2020","unstructured":"Feng, D., Haase-Sch\u00fctz, C., Rosenbaum, L., et al.: Deep multi-modal object detection and semantic segmentation for autonomous driving: datasets, methods, and challenges. IEEE Trans. Intell. Transp. Syst. 22(3), 1341\u20131360 (2020)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"3909_CR11","doi-asserted-by":"publisher","first-page":"638182","DOI":"10.3389\/fonc.2021.638182","volume":"11","author":"R Yang","year":"2021","unstructured":"Yang, R., Yu, Y.: Artificial convolutional neural network in object detection and semantic segmentation for medical imaging analysis. Front. Oncol. 11, 638182 (2021)","journal-title":"Front. Oncol."},{"key":"3909_CR12","doi-asserted-by":"crossref","unstructured":"Belem, R., Cruz, C., Pimentel, A., et al.: Automated video monitor screen extraction using semantic segmentation and CNN. In: 2020 IEEE International Conference on Consumer Electronics-Taiwan (ICCE-Taiwan), pp. 1\u20132. IEEE (2020)","DOI":"10.1109\/ICCE-Taiwan49838.2020.9258201"},{"key":"3909_CR13","doi-asserted-by":"crossref","unstructured":"Qiu, Y., Chang, C.S., Yan, J.L., et al.: Semantic segmentation of intracranial hemorrhages in head CT scans. In: 2019 IEEE 10th International Conference on Software Engineering and Service Science (ICSESS), pp. 112\u2013115. IEEE (2019)","DOI":"10.1109\/ICSESS47205.2019.9040733"},{"key":"3909_CR14","first-page":"2451","volume":"68","author":"J Amin","year":"2021","unstructured":"Amin, J., Sharif, M., Anjum, M.A., et al.: Diagnosis of COVID-19 infection using three-dimensional semantic segmentation and classification of computed tomography images. Comput. Mater. Contin. 68, 2451\u20132467 (2021)","journal-title":"Comput. Mater. Contin."},{"key":"3909_CR15","unstructured":"Goyal, M., Yap, M.H., Hassanpour, S.: Multi-class semantic segmentation of skin lesions via fully convolutional networks. arXiv preprint arXiv:1711.10449 (2017)"},{"key":"3909_CR16","doi-asserted-by":"publisher","first-page":"101619","DOI":"10.1016\/j.media.2019.101619","volume":"60","author":"K Wickstr\u00f8m","year":"2020","unstructured":"Wickstr\u00f8m, K., Kampffmeyer, M., Jenssen, R.: Uncertainty and interpretability in convolutional neural networks for semantic segmentation of colorectal polyps. Med. Image Anal. 60, 101619 (2020)","journal-title":"Med. Image Anal."},{"key":"3909_CR17","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1007\/978-1-4899-7641-3_9","volume-title":"Machine Learning Models and Algorithms for Big Data Classification: Thinking with Examples for Effective Learning","author":"S Suthaharan","year":"2016","unstructured":"Suthaharan, S., Suthaharan, S.: Support vector machine. In: Machine Learning Models and Algorithms for Big Data Classification: Thinking with Examples for Effective Learning, pp. 207\u2013235. Springer, Cham (2016)"},{"issue":"7","key":"3909_CR18","doi-asserted-by":"publisher","first-page":"3542","DOI":"10.1109\/TIP.2019.2905081","volume":"28","author":"B Kang","year":"2019","unstructured":"Kang, B., Nguyen, T.Q.: Random forest with learned representations for semantic segmentation. IEEE Trans. Image Process. 28(7), 3542\u20133555 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"3909_CR19","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1016\/j.patcog.2015.10.021","volume":"52","author":"D Rav\u00ec","year":"2016","unstructured":"Rav\u00ec, D., Bober, M., Farinella, G.M., et al.: Semantic segmentation of images exploiting DCT based features and random forest. Pattern Recogn. 52, 260\u2013273 (2016)","journal-title":"Pattern Recogn."},{"key":"3909_CR20","doi-asserted-by":"crossref","unstructured":"Fr\u00f6hlich, B., Rodner, E., Denzler, J.: Semantic segmentation with millions of features: integrating multiple cues in a combined random forest approach. In: Computer Vision\u2013ACCV 2012: 11th Asian Conference on Computer Vision, Daejeon, Korea, November 5\u20139, 2012, Revised Selected Papers, Part I 11, pp. 218\u2013231. Springer, Berlin Heidelberg (2013)","DOI":"10.1007\/978-3-642-37331-2_17"},{"key":"3909_CR21","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-Assisted Intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, October 5\u20139, 2015, Proceedings, Part III 18, pp. 234\u2013241. Springer (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"3909_CR22","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"3909_CR23","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"3909_CR24","doi-asserted-by":"publisher","first-page":"113120","DOI":"10.1016\/j.knosys.2025.113120","volume":"311","author":"Y Wang","year":"2025","unstructured":"Wang, Y., Zhang, Y., Zhang, L., et al.: A feature enhancement network based on image partitioning in a multi-branch encoder-decoder architecture. Knowl. Based Syst. 311, 113120 (2025)","journal-title":"Knowl. Based Syst."},{"key":"3909_CR25","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., et al.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"3909_CR26","unstructured":"Chen, L.C., Papandreou, G., Kokkinos, I., et al.: Semantic image segmentation with deep convolutional nets and fully connected crfs. arXiv preprint arXiv:1412.7062 (2014)"},{"key":"3909_CR27","unstructured":"Targ, S., Almeida, D., Lyman, K.: ResNet in ResNet: generalizing residual architectures. arXiv preprint arXiv:1603.08029 (2016)"},{"key":"3909_CR28","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., et al.: Pyramid scene parsing network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2881\u20132890 (2017)","DOI":"10.1109\/CVPR.2017.660"},{"issue":"4","key":"3909_CR29","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2017","unstructured":"Chen, L.C., Papandreou, G., Kokkinos, I., et al.: Deeplab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans. Pattern Anal. Mach. Intell. 40(4), 834\u2013848 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3909_CR30","doi-asserted-by":"crossref","unstructured":"Chen, L.C., Papandreou, G., Schroff, F., et al.: Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587 (2017)","DOI":"10.1007\/978-3-030-01234-2_49"},{"issue":"12","key":"3909_CR31","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan, V., Kendall, A., Cipolla, R.: SegNet: a deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 39(12), 2481\u20132495 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3909_CR32","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Rahman Siddiquee, M.M., Tajbakhsh, N., et al.: UNet++: a nested U-Net architecture for medical image segmentation. In: Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support: 4th International Workshop, DLMIA 2018, and 8th International Workshop, ML-CDS 2018, Held in Conjunction with MICCAI 2018, Granada, Spain, September 20, 2018, Proceedings 4, pp. 3\u201311. Springer (2018)","DOI":"10.1007\/978-3-030-00889-5_1"},{"key":"3909_CR33","doi-asserted-by":"crossref","unstructured":"Huang, H., Lin, L., Tong, R., et al.: UNet 3+: a full-scale connected UNet for medical image segmentation. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1055\u20131059. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053405"},{"key":"3909_CR34","doi-asserted-by":"publisher","first-page":"7195","DOI":"10.1109\/JBHI.2024.3447689","volume":"28","author":"R Agarwal","year":"2024","unstructured":"Agarwal, R., Chowdhury, A., Chatterjee, R.K., et al.: Deep quasi-recurrent self-attention with dual encoder-decoder in biomedical CT image segmentation. IEEE J. Biomed. Health Inform. 28, 7195 (2024)","journal-title":"IEEE J. Biomed. Health Inform."},{"key":"3909_CR35","doi-asserted-by":"publisher","first-page":"108464","DOI":"10.1016\/j.cmpb.2024.108464","volume":"257","author":"R Agarwal","year":"2024","unstructured":"Agarwal, R., Ghosal, P., Sadhu, A.K., et al.: Multi-scale dual-channel feature embedding decoder for biomedical image segmentation. Comput. Methods Programs Biomed. 257, 108464 (2024)","journal-title":"Comput. Methods Programs Biomed."},{"issue":"5","key":"3909_CR36","doi-asserted-by":"publisher","first-page":"749","DOI":"10.1109\/LGRS.2018.2802944","volume":"15","author":"Z Zhang","year":"2018","unstructured":"Zhang, Z., Liu, Q., Wang, Y.: Road extraction by deep residual U-Net. IEEE Geosci. Remote Sens. Lett. 15(5), 749\u2013753 (2018)","journal-title":"IEEE Geosci. Remote Sens. Lett."},{"key":"3909_CR37","doi-asserted-by":"crossref","unstructured":"Lin, G., Milan, A., Shen, C., et al.: RefineNet: multi-path refinement networks for high-resolution semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1925\u20131934 (2017)","DOI":"10.1109\/CVPR.2017.549"},{"issue":"1","key":"3909_CR38","doi-asserted-by":"publisher","first-page":"014006","DOI":"10.1117\/1.JMI.6.1.014006","volume":"6","author":"MZ Alom","year":"2019","unstructured":"Alom, M.Z., Yakopcic, C., Hasan, M., et al.: Recurrent residual U-Net for medical image segmentation. J. Med. Imaging 6(1), 014006\u2013014006 (2019)","journal-title":"J. Med. Imaging"},{"key":"3909_CR39","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1016\/j.isprsjprs.2020.01.013","volume":"162","author":"FI Diakogiannis","year":"2020","unstructured":"Diakogiannis, F.I., Waldner, F., Caccetta, P., et al.: ResUNet-a: a deep learning framework for semantic segmentation of remotely sensed data. ISPRS J. Photogramm. Remote Sens. 162, 94\u2013114 (2020)","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"issue":"3","key":"3909_CR40","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/s41095-022-0271-y","volume":"8","author":"MH Guo","year":"2022","unstructured":"Guo, M.H., Xu, T.X., Liu, J.J., et al.: Attention mechanisms in computer vision: a survey. Comput. Visual Media 8(3), 331\u2013368 (2022)","journal-title":"Comput. Visual Media"},{"key":"3909_CR41","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"3909_CR42","doi-asserted-by":"crossref","unstructured":"Wang, Q., Wu, B., Zhu, P., et al.: ECA-Net: efficient channel attention for deep convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11534\u201311542 (2020)","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"3909_CR43","doi-asserted-by":"publisher","first-page":"105889","DOI":"10.1016\/j.bspc.2023.105889","volume":"90","author":"J Zhang","year":"2024","unstructured":"Zhang, J., Luan, Z., Ni, L., et al.: MSDANet: a multi-scale dilation attention network for medical image segmentation. Biomed. Signal Process. Control 90, 105889 (2024)","journal-title":"Biomed. Signal Process. Control"},{"key":"3909_CR44","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.Y., et al.: CBAM: convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"3909_CR45","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1016\/j.patrec.2021.01.036","volume":"145","author":"K Trebing","year":"2021","unstructured":"Trebing, K., Sta\u01f9czyk, T., Mehrkanoon, S.: SmaAt-UNet: precipitation nowcasting using a small attention-UNet architecture. Pattern Recogn. Lett. 145, 178\u2013186 (2021)","journal-title":"Pattern Recogn. Lett."},{"key":"3909_CR46","doi-asserted-by":"publisher","first-page":"107798","DOI":"10.1016\/j.compbiomed.2023.107798","volume":"168","author":"R Wu","year":"2024","unstructured":"Wu, R., Lv, H., Liang, P., et al.: HSH-UNet: hybrid selective high order interactive U-shaped model for automated skin lesion segmentation. Comput. Biol. Med. 168, 107798 (2024)","journal-title":"Comput. Biol. Med."},{"key":"3909_CR47","unstructured":"Lan, L., Cai, P., Jiang, L., et al.: BRAU-Net++: U-shaped hybrid CNN-transformer network for medical image segmentation. arXiv preprint arXiv:2401.00722 (2024)"},{"key":"3909_CR48","doi-asserted-by":"publisher","first-page":"104038","DOI":"10.1016\/j.bspc.2022.104038","volume":"79","author":"H Song","year":"2023","unstructured":"Song, H., Wang, Y., Zeng, S., et al.: OAU-net: outlined attention U-net for biomedical image segmentation. Biomed. Signal Process. Control 79, 104038 (2023)","journal-title":"Biomed. Signal Process. Control"},{"key":"3909_CR49","doi-asserted-by":"publisher","first-page":"124179","DOI":"10.1016\/j.eswa.2024.124179","volume":"252","author":"Y Wang","year":"2024","unstructured":"Wang, Y., Zhang, Y., Zhang, L., et al.: Multi-bottleneck progressive propulsion network for medical image semantic segmentation with integrated macro-micro dual-stage feature enhancement and refinement. Expert Syst. Appl. 252, 124179 (2024)","journal-title":"Expert Syst. Appl."},{"key":"3909_CR50","doi-asserted-by":"crossref","unstructured":"Milletari, F., Navab, N., Ahmadi, S.A.: V-net: fully convolutional neural networks for volumetric medical image segmentation. In: 2016 4th International Conference on 3D Vision (3DV), pp. 565\u2013571. IEEE (2016)","DOI":"10.1109\/3DV.2016.79"},{"key":"3909_CR51","unstructured":"Xia, X., Kulis, B.: W-net: a deep model for fully unsupervised image segmentation. arXiv preprint arXiv:1711.08506 (2017)"},{"key":"3909_CR52","doi-asserted-by":"crossref","unstructured":"Qi, K., Yang, H., Li, C., et al.: X-net: brain stroke lesion segmentation based on depthwise separable convolution and long-range dependencies. In: Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2019: 22nd International Conference, Shenzhen, China, October 13\u201317, 2019, Proceedings, Part III 22, pp. 247\u2013255. Springer (2019)","DOI":"10.1007\/978-3-030-32248-9_28"},{"key":"3909_CR53","doi-asserted-by":"publisher","first-page":"107914","DOI":"10.1016\/j.cmpb.2023.107914","volume":"243","author":"Y Wang","year":"2024","unstructured":"Wang, Y., Yu, X., Yang, Y., et al.: A multi-branched semantic segmentation network based on twisted information sharing pattern for medical images. Comput. Methods Programs Biomed. 243, 107914 (2024)","journal-title":"Comput. Methods Programs Biomed."},{"key":"3909_CR54","doi-asserted-by":"crossref","unstructured":"\u00c7i\u00e7ek, \u00d6., Abdulkadir, A., Lienkamp, S.S., et al.: 3D U-Net: learning dense volumetric segmentation from sparse annotation. In: Medical Image Computing and Computer-Assisted Intervention\u2013MICCAI 2016: 19th International Conference, Athens, Greece, October 17\u201321, 2016, Proceedings, Part II 19, pp. 424\u2013432. Springer (2016)","DOI":"10.1007\/978-3-319-46723-8_49"},{"key":"3909_CR55","doi-asserted-by":"publisher","first-page":"107716","DOI":"10.1016\/j.ymssp.2021.107716","volume":"157","author":"Y Ye","year":"2021","unstructured":"Ye, Y., Huang, P., Sun, Y., et al.: MBSNet: a deep learning model for multibody dynamics simulation and its application to a vehicle-track system. Mech. Syst. Signal Process. 157, 107716 (2021)","journal-title":"Mech. Syst. Signal Process."},{"key":"3909_CR56","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., et al.: End-to-end object detection with transformers. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part I 16, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"3909_CR57","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et al.: An image is worth 16 \u00d7 16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"3909_CR58","doi-asserted-by":"publisher","first-page":"126098","DOI":"10.1016\/j.eswa.2024.126098","volume":"266","author":"Y Wang","year":"2024","unstructured":"Wang, Y., Wei, Y., Yu, X., et al.: A segmentation network for generalized lesion extraction with semantic fusion of transformer with value vector enhancement. Expert Syst. Appl. 266, 126098 (2024)","journal-title":"Expert Syst. Appl."},{"key":"3909_CR59","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"issue":"4","key":"3909_CR60","doi-asserted-by":"publisher","first-page":"2589","DOI":"10.1007\/s00371-023-02939-2","volume":"40","author":"D Yao","year":"2024","unstructured":"Yao, D., Shao, Y.: A data efficient transformer based on Swin transformer. Vis. Comput. 40(4), 2589\u20132598 (2024)","journal-title":"Vis. Comput."},{"key":"3909_CR61","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11042-024-19009-x","volume":"83","author":"J Yin","year":"2024","unstructured":"Yin, J., Chen, Y., Li, C., et al.: Swin-TransUper: Swin transformer-based UperNet for medical image segmentation. Multimed. Tools Appl. 83, 1\u201320 (2024)","journal-title":"Multimed. Tools Appl."},{"key":"3909_CR62","doi-asserted-by":"crossref","unstructured":"Xie, Y., Zhang, J., Shen, C., et al.: CoTr: efficiently bridging CNN and transformer for 3d medical image segmentation. In: Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2021: 24th International Conference, Strasbourg, France, September 27\u2013October 1, 2021, Proceedings, Part III 24, pp. 171\u2013180. Springer (2021)","DOI":"10.1007\/978-3-030-87199-4_16"},{"key":"3909_CR63","doi-asserted-by":"publisher","first-page":"102327","DOI":"10.1016\/j.media.2021.102327","volume":"76","author":"H Wu","year":"2022","unstructured":"Wu, H., Chen, S., Chen, G., et al.: FAT-Net: feature adaptive transformers for automated skin lesion segmentation. Med. Image Anal. 76, 102327 (2022)","journal-title":"Med. Image Anal."},{"key":"3909_CR64","doi-asserted-by":"publisher","first-page":"103856","DOI":"10.1016\/j.jvcir.2023.103856","volume":"95","author":"Y Wang","year":"2023","unstructured":"Wang, Y., Yu, X., Guo, X., et al.: A dual-decoding branch U-shaped semantic segmentation network combining transformer attention with decoder: DBUNet. J. Visual Commun. Image Represent. 95, 103856 (2023)","journal-title":"J. Visual Commun. Image Represent."},{"key":"3909_CR65","doi-asserted-by":"crossref","unstructured":"Wang, H., Xie, S., Lin, L., et al.: Mixed transformer U-Net for medical image segmentation. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 2390\u20132394. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746172"},{"key":"3909_CR66","unstructured":"Azad, R., Jia, Y., Aghdam, E.K., et al.: Enhancing medical image segmentation with TransCeption: a multi-scale feature fusion approach. arXiv preprint arXiv:2301.10847 (2023)"},{"key":"3909_CR67","doi-asserted-by":"crossref","unstructured":"Cao, H., Wang, Y., Chen, J., et al.: Swin-UNet: UNet-like pure transformer for medical image segmentation. In: European Conference on Computer Vision, pp. 205\u2013218. Springer, Cham (2022)","DOI":"10.1007\/978-3-031-25066-8_9"},{"key":"3909_CR68","doi-asserted-by":"crossref","unstructured":"Fan, C.M., Liu, T.J., Liu, K.H.: SUNet: Swin transformer UNet for image denoising. In: 2022 IEEE International Symposium on Circuits and Systems (ISCAS), pp. 2333\u20132337. IEEE (2022)","DOI":"10.1109\/ISCAS48785.2022.9937486"},{"key":"3909_CR69","doi-asserted-by":"crossref","unstructured":"Heidari, M., Kazerouni, A., Soltany, M., et al.: Hiformer: hierarchical multi-scale representations using transformers for medical image segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6202\u20136212 (2023)","DOI":"10.1109\/WACV56688.2023.00614"},{"key":"3909_CR70","first-page":"1","volume":"41","author":"R Feng","year":"2024","unstructured":"Feng, R., Wang, Y., Xue, J., et al.: CLAC-Net: a composite medical image segmentation framework using self-attention and cross-layer asymmetric connections. Visual Comput. 41, 1\u201331 (2024)","journal-title":"Visual Comput."},{"key":"3909_CR71","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., et al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03909-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03909-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03909-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T14:37:44Z","timestamp":1757169464000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03909-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,11]]},"references-count":71,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["3909"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03909-6","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,11]]},"assertion":[{"value":"29 March 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}]}}