{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,9]],"date-time":"2025-04-09T04:25:32Z","timestamp":1744172732654,"version":"3.40.3"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,3,10]],"date-time":"2025-03-10T00:00:00Z","timestamp":1741564800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,10]],"date-time":"2025-03-10T00:00:00Z","timestamp":1741564800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62167006","62167006","62167006","62167006","62167006","62167006","62167006"],"award-info":[{"award-number":["62167006","62167006","62167006","62167006","62167006","62167006","62167006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the Key project of China National Social Science Foundation","award":["20AXW009","20AXW009","20AXW009","20AXW009","20AXW009","20AXW009","20AXW009"],"award-info":[{"award-number":["20AXW009","20AXW009","20AXW009","20AXW009","20AXW009","20AXW009","20AXW009"]}]},{"name":"the Jiangxi Province Natural Science Foundation project","award":["20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017"],"award-info":[{"award-number":["20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017","20212BAB202017"]}]},{"name":"the Jiangxi Provincial Science and Technology Innovation Base Program Project - Jiangxi Provincial Key Laboratory of Intelligent Information Processing and Affective Computing","award":["20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021"],"award-info":[{"award-number":["20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021","20242BCC32021"]}]},{"name":"Academic and Technical Leaders (Leading Talents) Project of Major Disciplines in Jiangxi Province","award":["20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047"],"award-info":[{"award-number":["20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047","20213BCJL22047"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11760-025-03929-8","type":"journal-article","created":{"date-parts":[[2025,3,10]],"date-time":"2025-03-10T08:19:04Z","timestamp":1741594744000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Semantic injection and multi-scale cross-axis attention-based portrait segmentation model"],"prefix":"10.1007","volume":"19","author":[{"given":"Yan","family":"Cheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanying","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zou","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guowei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gongcheng","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaqi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxiao","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,10]]},"reference":[{"key":"3929_CR1","doi-asserted-by":"crossref","unstructured":"Shen, X., Hertzmann, A., Jia, J., Paris, S., Price, B., Shechtman, E., Sachs, I.: Automatic portrait segmentation for image stylization. In: Computer Graphics Forum, vol. 35, pp. 93\u2013102 (2016). Wiley Online Library","DOI":"10.1111\/cgf.12814"},{"issue":"6","key":"3929_CR2","first-page":"219","volume":"58","author":"C Wei","year":"2022","unstructured":"Wei, C., Dong, H., Xu, X.: Attribute editable personimage synthesis based on spatial transformation. Comput. Eng. Appl. 58(6), 219\u2013226 (2022)","journal-title":"Comput. Eng. Appl."},{"key":"3929_CR3","unstructured":"Xingyue, D.U., Hongwei, D., Zhen, Y.: Face region segmentation method based on deep network. Computer Engineering and Applications (2019)"},{"key":"3929_CR4","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural Inf. Process. Syst.25 (2012)"},{"key":"3929_CR5","doi-asserted-by":"crossref","unstructured":"Zhang, X., Zhou, X., Lin, M., Sun, J.: Shufflenet: An extremely efficient convolutional neural network for mobile devices. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6848\u20136856 (2018)","DOI":"10.1109\/CVPR.2018.00716"},{"key":"3929_CR6","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Zheng, H.-T., Sun, J.: Shufflenet v2: Practical guidelines for efficient cnn architecture design. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 116\u2013131 (2018)","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"3929_CR7","unstructured":"Tan, M., Le, Q.: Efficientnet: Rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114 (2019). PMLR"},{"key":"3929_CR8","unstructured":"Tan, M., Le, Q.: Efficientnetv2: Smaller models and faster training. In: International Conference on Machine Learning, pp. 10096\u201310106 (2021). PMLR"},{"key":"3929_CR9","unstructured":"Howard, A.G.: Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017)"},{"key":"3929_CR10","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv2: Inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"3929_CR11","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V., et al.: Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1314\u20131324 (2019)","DOI":"10.1109\/ICCV.2019.00140"},{"key":"3929_CR12","doi-asserted-by":"crossref","unstructured":"Zhou, D., Hou, Q., Chen, Y., Feng, J., Yan, S.: Rethinking bottleneck structure for efficient mobile network design. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part III 16, pp. 680\u2013697 (2020). Springer","DOI":"10.1007\/978-3-030-58580-8_40"},{"key":"3929_CR13","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: Transformers for image recognition at scale. ICLR (2021)"},{"key":"3929_CR14","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: Segformer: simple and efficient design for semantic segmentation with transformers. Adv. Neural. Inf. Process. Syst. 34, 12077\u201312090 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3929_CR15","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"3929_CR16","unstructured":"Huang, Z., Ben, Y., Luo, G., Cheng, P., Yu, G., Fu, B.: Shuffle transformer: Rethinking spatial shuffle for vision transformer. arXiv preprint arXiv:2106.03650 (2021)"},{"key":"3929_CR17","unstructured":"Mehta, S., Rastegari, M.: Mobilevit: Light-weight, general-purpose, and mobile-friendly vision transformer. In: International Conference on Learning Representations (2022)"},{"key":"3929_CR18","doi-asserted-by":"crossref","unstructured":"Wang, H., Zhu, Y., Green, B., Adam, H., Yuille, A., Chen, L.-C.: Axial-deeplab: Stand-alone axial-attention for panoptic segmentation. In: European Conference on Computer Vision, pp. 108\u2013126 (2020). Springer","DOI":"10.1007\/978-3-030-58548-8_7"},{"key":"3929_CR19","doi-asserted-by":"crossref","unstructured":"Valanarasu, J.M.J., Oza, P., Hacihaliloglu, I., Patel, V.M.: Medical transformer: Gated axial-attention for medical image segmentation. In: Medical Image Computing and Computer Assisted intervention\u2013MICCAI 2021: 24th International Conference, Strasbourg, France, September 27 Oct 1, 2021, Proceedings, Part I 24, pp. 36\u201346 (2021). Springer","DOI":"10.1007\/978-3-030-87193-2_4"},{"key":"3929_CR20","doi-asserted-by":"crossref","unstructured":"Zhang, W., Huang, Z., Luo, G., Chen, T., Wang, X., Liu, W., Yu, G., Shen, C.: Topformer: Token pyramid transformer for mobile semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12083\u201312093 (2022)","DOI":"10.1109\/CVPR52688.2022.01177"},{"key":"3929_CR21","unstructured":"Shao, H., Zeng, Q., Hou, Q., Yang, J.: Mcanet: Medical image segmentation with multi-scale cross-axis attention. arXiv preprint arXiv:2312.08866 (2023)"},{"key":"3929_CR22","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1016\/j.neucom.2019.02.003","volume":"338","author":"F Lateef","year":"2019","unstructured":"Lateef, F., Ruichek, Y.: Survey on semantic segmentation using deep learning techniques. Neurocomputing 338, 321\u2013348 (2019)","journal-title":"Neurocomputing"},{"key":"3929_CR23","doi-asserted-by":"crossref","unstructured":"Ding, M., Lian, X., Yang, L., Wang, P., Jin, X., Lu, Z., Luo, P.: Hr-nas: Searching efficient high-resolution neural architectures with lightweight transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2982\u20132992 (2021)","DOI":"10.1109\/CVPR46437.2021.00300"},{"key":"3929_CR24","doi-asserted-by":"crossref","unstructured":"Huang, D., Wu, D., Liu, J., Lv, Y.: Ddcnet: A lightweight network with variable receptive field for real-time portrait segmentation in complex environment. In: Computer Graphics International Conference, pp. 465\u2013476 (2022). Springer","DOI":"10.1007\/978-3-031-23473-6_36"},{"key":"3929_CR25","doi-asserted-by":"crossref","unstructured":"Yu, C., Wang, J., Peng, C., Gao, C., Yu, G., Sang, N.: Bisenet: Bilateral segmentation network for real-time semantic segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 325\u2013341 (2018)","DOI":"10.1007\/978-3-030-01261-8_20"},{"key":"3929_CR26","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"3929_CR27","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-assisted intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, Oct 5-9, 2015, Proceedings, Part III 18, pp. 234\u2013241 (2015). Springer","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"3929_CR28","doi-asserted-by":"crossref","unstructured":"Wu, K., Zhang, J., Peng, H., Liu, M., Xiao, B., Fu, J., Yuan, L.: Tinyvit: Fast pretraining distillation for small vision transformers. In: European Conference on Computer Vision, pp. 68\u201385 (2022). Springer","DOI":"10.1007\/978-3-031-19803-8_5"},{"key":"3929_CR29","doi-asserted-by":"crossref","unstructured":"Graham, B., El-Nouby, A., Touvron, H., Stock, P., Joulin, A., J\u00e9gou, H., Douze, M.: Levit: a vision transformer in convnet\u2019s clothing for faster inference. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12259\u201312269 (2021)","DOI":"10.1109\/ICCV48922.2021.01204"},{"key":"3929_CR30","first-page":"1140","volume":"35","author":"MH Guo","year":"2022","unstructured":"Guo, M.H., Lu, C.Z., Hou, Q., et al.: Segnext: Rethinking convolutional attention design for semantic segmentation[J]. Adv Neural Info Process Syst 35, 1140\u20131156 (2022)","journal-title":"Adv Neural Info Process Syst"},{"key":"3929_CR31","doi-asserted-by":"crossref","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H.: Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 801\u2013818 (2018)","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"3929_CR32","doi-asserted-by":"crossref","unstructured":"Yu, W., Luo, M., Zhou, P., Si, C., Zhou, Y., Wang, X., Feng, J., Yan, S.: Metaformer is actually what you need for vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10819\u201310829 (2022)","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"3929_CR33","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J.: Pyramid scene parsing network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2881\u20132890 (2017)","DOI":"10.1109\/CVPR.2017.660"},{"key":"3929_CR34","unstructured":"Wan, Q., Huang, Z., Lu, J., Yu, G., Zhang, L.: Seaformer: squeeze-enhanced axial transformer for mobile semantic segmentation. In: International Conference on Learning Representations (ICLR) (2023)"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-03929-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-03929-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-03929-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,8]],"date-time":"2025-04-08T20:11:26Z","timestamp":1744143086000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-03929-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,10]]},"references-count":34,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["3929"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-03929-8","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"type":"print","value":"1863-1703"},{"type":"electronic","value":"1863-1711"}],"subject":[],"published":{"date-parts":[[2025,3,10]]},"assertion":[{"value":"31 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 February 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 March 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"384"}}