{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:02:37Z","timestamp":1784894557047,"version":"3.55.0"},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Shenzhen University 2035 Program for Excellent Research","award":["00000224"],"award-info":[{"award-number":["00000224"]}]},{"name":"National Major Scientific Instruments and Equipments Development Project of National Natural Science Foundation of China","award":["62327808"],"award-info":[{"award-number":["62327808"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China Project","doi-asserted-by":"crossref","award":["62403326"],"award-info":[{"award-number":["62403326"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Guangdong Major Project of Basic Research","award":["2023B0303000009"],"award-info":[{"award-number":["2023B0303000009"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00530-026-02493-6","type":"journal-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T14:42:55Z","timestamp":1783780975000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["AS-FPN: an asymmetric semantic-preserving feature pyramid network for efficient semantic segmentation"],"prefix":"10.1007","volume":"32","author":[{"given":"Jiawei","family":"Pan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deyu","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zongze","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanyun","family":"Qu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weixiang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,11]]},"reference":[{"key":"2493_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102608","volume":"114","author":"KK Brar","year":"2025","unstructured":"Brar, K.K., Goyal, B., Dogra, A., Mustafa, M.A., Majumdar, R., Alkhayyat, A., Kukreja, V.: Image segmentation review: theoretical background and recent advances. Inf. Fusion 114, 102608 (2025). https:\/\/doi.org\/10.1016\/j.inffus.2024.102608","journal-title":"Inf. Fusion"},{"issue":"2","key":"2493_CR2","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s00530-024-01262-7","volume":"30","author":"Z Liang","year":"2024","unstructured":"Liang, Z., Dong, W., Zhang, B.: A dual-branch hybrid network of cnn and transformer with adaptive keyframe scheduling for video semantic segmentation. Multimedia Syst. 30(2), 67 (2024). https:\/\/doi.org\/10.1007\/s00530-024-01262-7","journal-title":"Multimedia Syst."},{"issue":"2","key":"2493_CR3","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1007\/s00530-026-02215-y","volume":"32","author":"X Lu","year":"2026","unstructured":"Lu, X., Zhou, C., Lu, Y., Zhou, Z.: Dt-net: a hybrid framework of dcnn and transformer for medical image segmentation. Multimedia Syst. 32(2), 142 (2026). https:\/\/doi.org\/10.1007\/s00530-026-02215-y","journal-title":"Multimedia Syst."},{"key":"2493_CR4","doi-asserted-by":"publisher","unstructured":"Xiao, T., Liu, Y., Zhou, B., Jiang, Y., Sun, J.: Unified perceptual parsing for scene understanding. In: Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany, pp. 418\u2013434 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01228-1_26","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"2493_CR5","doi-asserted-by":"publisher","unstructured":"Cheng, B., Misra, I., Schwing, A.G., Kirillov, A., Girdhar, R.: Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), New Orleans, USA, pp. 1290\u20131299 (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.00135","DOI":"10.1109\/cvpr52688.2022.00135"},{"issue":"6","key":"2493_CR6","doi-asserted-by":"publisher","first-page":"342","DOI":"10.1007\/s00530-024-01580-w","volume":"30","author":"Y Liu","year":"2024","unstructured":"Liu, Y., Xie, L., Ye, W.: Edb-diff: a edgedevice based diffusion network for brain tumor image segmentation. Multimedia Syst. 30(6), 342 (2024). https:\/\/doi.org\/10.1007\/s00530-024-01580-w","journal-title":"Multimedia Syst."},{"key":"2493_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/s10844-025-00984-y","author":"S Polimena","year":"2025","unstructured":"Polimena, S., Pio, G., Attolico, G., Ceci, M.: Handling complex backgrounds and light perturbations for enhancing learning tasks from images of vegetables. J. Intell. Inf. Syst. (2025). https:\/\/doi.org\/10.1007\/s10844-025-00984-y","journal-title":"J. Intell. Inf. Syst."},{"key":"2493_CR8","doi-asserted-by":"publisher","first-page":"12077","DOI":"10.48550\/arXiv.2105.15203","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: Segformer: Simple and efficient design for semantic segmentation with transformers. Adv. Neural. Inf. Process. Syst. 34, 12077\u201312090 (2021). https:\/\/doi.org\/10.48550\/arXiv.2105.15203","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2493_CR9","doi-asserted-by":"publisher","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T., Xie, S.: A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), New Orleans, USA, 11976\u201311986 (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01167","DOI":"10.1109\/cvpr52688.2022.01167"},{"key":"2493_CR10","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.48550\/arXiv.2209.08575","volume":"35","author":"M-H Guo","year":"2022","unstructured":"Guo, M.-H., Lu, C.-Z., Hou, Q., Liu, Z., Cheng, M.-M., Hu, S.-M.: Segnext: Rethinking convolutional attention design for semantic segmentation. Adv. Neural. Inf. Process. Syst. 35, 1140\u20131156 (2022). https:\/\/doi.org\/10.48550\/arXiv.2209.08575","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2493_CR11","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Honolulu, HI, USA, pp. 2117\u20132125 (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.106","DOI":"10.1109\/cvpr.2017.106"},{"key":"2493_CR12","doi-asserted-by":"publisher","unstructured":"Tan, M., Pang, R., Le, Q.V.: Efficientdet: Scalable and efficient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), Seattle, USA, pp. 10781\u201310790 (2020). https:\/\/doi.org\/10.1109\/cvpr42600.2020.01079","DOI":"10.1109\/cvpr42600.2020.01079"},{"key":"2493_CR13","doi-asserted-by":"publisher","unstructured":"Zhou, B., Zhao, H., Puig, X., Xiao, T., Fidler, S., Barriuso, A., Torralba, A.: Semantic understanding of scenes through the ade20k dataset. Int. J. Comput. Vis. 127, 302\u2013321 (2019). https:\/\/doi.org\/10.1007\/s11263-018-1140-0","DOI":"10.1007\/s11263-018-1140-0"},{"key":"2493_CR14","doi-asserted-by":"publisher","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., Schiele, B.: The cityscapes dataset for semantic urban scene understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Las Vegas, NV, USA, pp. 3213\u20133223 (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.350","DOI":"10.1109\/cvpr.2016.350"},{"key":"2493_CR15","doi-asserted-by":"publisher","unstructured":"Caesar, H., Uijlings, J., Ferrari, V.: Coco-stuff: Thing and stuff classes in context. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Salt Lake City, UT, USA, pp. 1209\u20131218 (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00132","DOI":"10.1109\/cvpr.2018.00132"},{"issue":"6","key":"2493_CR16","doi-asserted-by":"publisher","first-page":"3645","DOI":"10.1007\/s11263-025-02345-2","volume":"133","author":"Q Wan","year":"2025","unstructured":"Wan, Q., Huang, Z., Lu, J., Yu, G., Zhang, L.: Seaformer++: squeeze-enhanced axial transformer for mobile visual recognition. Int. J. Comput. Vis. 133(6), 3645\u20133666 (2025). https:\/\/doi.org\/10.1007\/s11263-025-02345-2","journal-title":"Int. J. Comput. Vis."},{"issue":"4","key":"2493_CR17","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1109\/tpami.2016.2572683","volume":"39","author":"E Shelhamer","year":"2016","unstructured":"Shelhamer, E., Long, J., Darrell, T.: Fully convolutional networks for semantic segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 39(4), 640\u2013651 (2016). https:\/\/doi.org\/10.1109\/tpami.2016.2572683","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2493_CR18","doi-asserted-by":"publisher","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J.: Pyramid scene parsing network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Honolulu, HI, USA, pp. 2881\u20132890 (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.660","DOI":"10.1109\/cvpr.2017.660"},{"key":"2493_CR19","doi-asserted-by":"publisher","unstructured":"Chen, L.-C., Zhu, Y., Papandreou, G., Schroff, F., Adam, H.: Encoder-decoder with atrous separable convolution for semantic image segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany, pp. 801\u2013818 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_49","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"2493_CR20","doi-asserted-by":"publisher","unstructured":"Yu, C., Wang, J., Peng, C., Gao, C., Yu, G., Sang, N.: Bisenet: bilateral segmentation network for real-time semantic segmentation. In: Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany, pp. 325\u2013341 (2018).https:\/\/doi.org\/10.1007\/978-3-030-01261-8_20","DOI":"10.1007\/978-3-030-01261-8_20"},{"issue":"3","key":"2493_CR21","doi-asserted-by":"publisher","first-page":"3448","DOI":"10.1109\/TITS.2022.3228042","volume":"24","author":"H Pan","year":"2023","unstructured":"Pan, H., Hong, Y., Sun, W., Jia, Y.: Deep dual-resolution networks for real-time and accurate semantic segmentation of traffic scenes. IEEE Trans. Intell. Transp. Syst. 24(3), 3448\u20133460 (2023). https:\/\/doi.org\/10.1109\/TITS.2022.3228042","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"2493_CR22","doi-asserted-by":"publisher","unstructured":"Xu, J., Xiong, Z., Bhattacharyya, S.P.: Pidnet: A real-time semantic segmentation network inspired by pid controllers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), Montreal, Canada, pp. 19529\u201319539 (2023). https:\/\/doi.org\/10.1109\/cvpr52729.2023.01871","DOI":"10.1109\/cvpr52729.2023.01871"},{"key":"2493_CR23","doi-asserted-by":"publisher","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V., Le, Q.V., Adam, H.: Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), Long Beach, CA, USA, pp. 1314\u20131324 (2019). https:\/\/doi.org\/10.1109\/iccv.2019.00140","DOI":"10.1109\/iccv.2019.00140"},{"key":"2493_CR24","doi-asserted-by":"publisher","unstructured":"Zhang, W., Huang, Z., Luo, G., Chen, T., Wang, X., Liu, W., Yu, G., Shen, C.: Topformer: Token pyramid transformer for mobile semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), New Orleans, USA, pp. 12083\u201312093 (2022). https:\/\/doi.org\/10.1109\/cvpr52688.2022.01177","DOI":"10.1109\/cvpr52688.2022.01177"},{"key":"2493_CR25","doi-asserted-by":"publisher","unstructured":"Liu, S., Qi, L., Qin, H., Shi, J., Jia, J.: Path aggregation network for instance segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Salt Lake City, UT, USA, pp. 8759\u20138768 (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00913","DOI":"10.1109\/cvpr.2018.00913"},{"key":"2493_CR26","doi-asserted-by":"publisher","unstructured":"Ghiasi, G., Lin, T.-Y., Le, Q.V.: Nas-fpn: Learning scalable feature pyramid architecture for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), New Orleans, USA, pp. 7036\u20137045 (2019). https:\/\/doi.org\/10.1109\/cvpr.2019.00720","DOI":"10.1109\/cvpr.2019.00720"},{"key":"2493_CR27","doi-asserted-by":"publisher","unstructured":"Yu, F., Koltun, V.: Multi-scale context aggregation by dilated convolutions (2015). arXiv preprint arXiv:1511.07122. https:\/\/doi.org\/10.48550\/arXiv.1511.07122","DOI":"10.48550\/arXiv.1511.07122"},{"key":"2493_CR28","doi-asserted-by":"publisher","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Salt Lake City, UT, USA, pp. 7794\u20137803 (2018). https:\/\/doi.org\/10.1109\/cvpr.2017.623","DOI":"10.1109\/cvpr.2017.623"},{"key":"2493_CR29","doi-asserted-by":"publisher","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Salt Lake City, UT, USA, pp. 7132\u20137141 (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00745","DOI":"10.1109\/cvpr.2018.00745"},{"key":"2493_CR30","doi-asserted-by":"publisher","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S.: Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany, pp. 3\u201319 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_1","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"2493_CR31","doi-asserted-by":"publisher","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), Montreal, Canada, pp. 10012\u201310022 (2021).https:\/\/doi.org\/10.1109\/iccv48922.2021.00986","DOI":"10.1109\/iccv48922.2021.00986"},{"key":"2493_CR32","doi-asserted-by":"publisher","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., etal An image is worth 16x16 words: transformers for image recognition at scale (2020). arXiv preprint arXiv:2010.11929. https:\/\/doi.org\/10.48550\/arXiv.2010.11929","DOI":"10.48550\/arXiv.2010.11929"},{"key":"2493_CR33","unstructured":"Contributors, M.: MMSegmentation: OpenMMLab Semantic Segmentation Toolbox and Benchmark (2020). https:\/\/github.com\/open-mmlab\/mmsegmentation"},{"key":"2493_CR34","doi-asserted-by":"publisher","first-page":"3051","DOI":"10.1007\/s11263-021-01515-2","volume":"129","author":"C Yu","year":"2021","unstructured":"Yu, C., Gao, C., Wang, J., Yu, G., Shen, C., Sang, N.: Bisenet v2: bilateral network with guided aggregation for real-time semantic segmentation. Int. J. Comput. Vis. 129, 3051\u20133068 (2021). https:\/\/doi.org\/10.1007\/s11263-021-01515-2","journal-title":"Int. J. Comput. Vis."},{"key":"2493_CR35","doi-asserted-by":"publisher","unstructured":"Wan, Q., Huang, Z., Lu, J., Yu, G., Zhang, L.: Seaformer: Squeeze-enhanced axial transformer for mobile semantic segmentation. In: The Eleventh International Conference on Learning Representations (ICLR), Kigali, Rwanda (2023). https:\/\/doi.org\/10.48550\/arXiv.2301.13156","DOI":"10.48550\/arXiv.2301.13156"},{"issue":"2","key":"2493_CR36","doi-asserted-by":"publisher","first-page":"2400","DOI":"10.1109\/tpami.2022.3162528","volume":"45","author":"T Verelst","year":"2022","unstructured":"Verelst, T., Tuytelaars, T.: Segblocks: Block-based dynamic resolution networks for real-time segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 45(2), 2400\u20132411 (2022). https:\/\/doi.org\/10.1109\/tpami.2022.3162528","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02493-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02493-6","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02493-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:30:50Z","timestamp":1784892650000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02493-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":36,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2493"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02493-6","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"14 February 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"419"}}