{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:03:38Z","timestamp":1784894618886,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Nantong Science and Technology Program","award":["MSZ2024059"],"award-info":[{"award-number":["MSZ2024059"]}]},{"name":"Research Projects of the NIT External Faculty Doctoral Studio","award":["WP202534"],"award-info":[{"award-number":["WP202534"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00371-026-04619-3","type":"journal-article","created":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T09:08:22Z","timestamp":1783156102000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Hybrid token learning with bidirectional attention for few-shot semantic segmentation"],"prefix":"10.1007","volume":"42","author":[{"given":"Liting","family":"Lei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujie","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yadang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianlin","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,4]]},"reference":[{"key":"4619_CR1","doi-asserted-by":"crossref","unstructured":"Long, J., Shelhamer, E., Darrell, T.: Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3431\u20133440 (2015)","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"4619_CR2","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: Convolutional networks for biomedical image segmentation. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 234\u2013241, Springer (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"issue":"4","key":"4619_CR3","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"L-C Chen","year":"2017","unstructured":"Chen, L.-C., Papandreou, G., Kokkinos, I., Murphy, K., Yuille, A.L.: Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans. Pattern Anal. Mach. Intell. 40(4), 834\u2013848 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4619_CR4","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., Jia, J.: Pyramid scene parsing network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2881\u20132890 (2017)","DOI":"10.1109\/CVPR.2017.660"},{"issue":"7","key":"4619_CR5","first-page":"8912","volume":"45","author":"Y Liu","year":"2023","unstructured":"Liu, Y., Liu, N., Cao, Q., Yao, X., Han, J.: Learning non-target knowledge for few-shot semantic segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 45(7), 8912\u20138925 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4619_CR6","unstructured":"Cao, L., Guo, Y., Yuan, Y., Jin, Q.: Prototype as query for few-shot semantic segmentation. arXiv:2211.14764 (2022)"},{"key":"4619_CR7","doi-asserted-by":"crossref","unstructured":"Liu, Y., Liu, N., Yao, X., Han, J.: Intermediate prototype mining transformer for few-shot semantic segmentation. In: Adv. Neural Inf. Process. Syst.,35:38020\u201338031 (2022)","DOI":"10.52202\/068431-2755"},{"key":"4619_CR8","doi-asserted-by":"crossref","unstructured":"Yun, J., Ak\u00e7akaya, M.: Generative model-based fusion for improved few-shot semantic segmentation of infrared images. In: Proceedings IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 5479\u20135488 (2025)","DOI":"10.1109\/WACV61041.2025.00535"},{"key":"4619_CR9","unstructured":"Immanuel, S.A., Cho, W., Heo, J., et al.: Tackling few-shot segmentation in remote sensing via inpainting diffusion model. arXiv:2503.03785 (2025)"},{"key":"4619_CR10","doi-asserted-by":"publisher","first-page":"1432","DOI":"10.1109\/TIP.2024.3364056","volume":"33","author":"Y Chen","year":"2024","unstructured":"Chen, Y., Jiang, R., Zheng, Y., Sheng, B., Yang, Z.-X., Wu, E.: Dual branch multi-level semantic learning for few-shot segmentation. IEEE Trans. Image Process. 33, 1432\u20131447 (2024). (CCF-A)","journal-title":"IEEE Trans. Image Process."},{"issue":"7","key":"4619_CR11","doi-asserted-by":"publisher","first-page":"5560","DOI":"10.1109\/TCSVT.2024.3358679","volume":"34","author":"Z Chang","year":"2024","unstructured":"Chang, Z., Gao, X., Li, N., Zhou, H., Lu, Y.: Drnet: Disentanglement and recombination network for few-shot semantic segmentation. IEEE Trans. Circuits Syst. Video Technol. 34(7), 5560\u20135574 (2024). https:\/\/doi.org\/10.1109\/TCSVT.2024.3358679","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4619_CR12","doi-asserted-by":"crossref","unstructured":"Bao, X., Qin, J., Sun, S., et al.: Relevant intrinsic feature enhancement network for few-shot semantic segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 765\u2013773 (2024)","DOI":"10.1609\/aaai.v38i2.27834"},{"key":"4619_CR13","doi-asserted-by":"crossref","unstructured":"Wang, J., Zhang, B., Pang, J., et al.: Rethinking prior information generation with clip for few-shot segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 3941\u20133951 (2024)","DOI":"10.1109\/CVPR52733.2024.00378"},{"key":"4619_CR14","doi-asserted-by":"crossref","unstructured":"Sakurai, K., Shimizu, R., Goto, M.: Vision and language reference prompt into sam for few-shot segmentation. arXiv:2502.00719 (2025)","DOI":"10.3390\/jimaging12040143"},{"key":"4619_CR15","doi-asserted-by":"crossref","unstructured":"Zheng, S., Lu, J., Zhao, H., Zhu, X., Luo, Z., Wang, Y., Fu, Y., Feng, J., Xiang, T., Torr, P.H., et al.: Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6881\u20136890 (2021)","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"4619_CR16","first-page":"12077","volume":"34","author":"E Xie","year":"2021","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: Segformer: Simple and efficient design for semantic segmentation with transformers. Adv. Neural. Inf. Process. Syst. 34, 12077\u201312090 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4619_CR17","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"4619_CR18","unstructured":"Chen, L.-C., Papandreou, G., Schroff, F., Adam, H.: Rethinking atrous convolution for semantic image segmentation. arXiv:1706.05587 (2017)"},{"key":"4619_CR19","doi-asserted-by":"crossref","unstructured":"Wang, K., Liew, J.H., Zou, Y., et al.: Panet: Few-shot image semantic segmentation with prototype alignment. In: proceedings of the IEEE\/CVF international conference on computer vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00929"},{"key":"4619_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Y., Zhang, X., Zhang, S., He, X.: Part-aware prototype network for few-shot semantic segmentation. In: Proceedings European conference on computer vision (ECCV) (2020)","DOI":"10.1007\/978-3-030-58545-7_9"},{"key":"4619_CR21","doi-asserted-by":"crossref","unstructured":"Li, G., Jampani, V., Sevilla-Lara, L., et al.: Adaptive prototype learning and allocation for few-shot segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8334\u20138343 (2021)","DOI":"10.1109\/CVPR46437.2021.00823"},{"issue":"9","key":"4619_CR22","doi-asserted-by":"publisher","first-page":"3855","DOI":"10.1109\/TCYB.2020.2992433","volume":"50","author":"X Zhang","year":"2020","unstructured":"Zhang, X., Wei, Y., Yang, Y., Huang, T.S.: Sg-one: Similarity guidance network for one-shot semantic segmentation. IEEE Trans. Cybern. 50(9), 3855\u20133865 (2020). https:\/\/doi.org\/10.1109\/TCYB.2020.2992433","journal-title":"IEEE Trans. Cybern."},{"key":"4619_CR23","unstructured":"Zhang, C., Lin, G., Liu, F., et al.: Canet: Class-agnostic segmentation networks with iterative refinement and attentive few-shot learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022)"},{"issue":"8","key":"4619_CR24","first-page":"2785","volume":"43","author":"Z Tian","year":"2021","unstructured":"Tian, Z., Zhao, H., Shu, M., et al.: Prior guided feature enrichment network for few-shot segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 43(8), 2785\u20132797 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4619_CR25","doi-asserted-by":"crossref","unstructured":"Min, J., Kang, D., Cho, M.: Hypercorrelation squeeze for few-shot segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision pp. 6941\u20136952 (2021)","DOI":"10.1109\/ICCV48922.2021.00686"},{"key":"4619_CR26","first-page":"21984","volume":"34","author":"G Zhang","year":"2021","unstructured":"Zhang, G., Kang, G., Yang, Y., Wei, Y.: Few-shot segmentation via cycle-consistent transformer. Adv. Neural. Inf. Process. Syst. 34, 21984\u201321996 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"9","key":"4619_CR27","doi-asserted-by":"publisher","first-page":"10669","DOI":"10.1109\/TPAMI.2023.3265865","volume":"45","author":"C Lang","year":"2023","unstructured":"Lang, C., Cheng, G., Tu, B., Li, C., Han, J.: Base and meta: A new perspective on few-shot segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 45(9), 10669\u201310686 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4619_CR28","doi-asserted-by":"crossref","unstructured":"Shi, X., Wei, D., Zhang, Y., Lu, D., Ning, M., Chen, J., Ma, K., Zheng, Y.: Dense cross-query-and-support attention weighted mask aggregation for few-shot segmentation. In: European Conference on Computer Vision pp. 151\u2013168, Springer (2022)","DOI":"10.1007\/978-3-031-20044-1_9"},{"key":"4619_CR29","doi-asserted-by":"crossref","unstructured":"Yang, Y., Chen, Q., Feng, Y., Huang, T.: Mianet: Aggregating unbiased instance and general information for few-shot semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition pp. 7131\u20137140 (2023)","DOI":"10.1109\/CVPR52729.2023.00689"},{"key":"4619_CR30","doi-asserted-by":"crossref","unstructured":"Fan, Q., Pei, W., Tai, Y.-W., Tang, C.-K.: Self-support few-shot semantic segmentation. In: European Conference on Computer Vision pp. 701\u2013719, Springer (2022)","DOI":"10.1007\/978-3-031-19800-7_41"},{"key":"4619_CR31","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Adv. neural inf. proc. syst. 30 (2017)"},{"key":"4619_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"4619_CR33","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., et al.: End-to-end object detection with transformers. In: European Conference on Computer Vision (ECCV), pp. 213\u2013229 (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"4619_CR34","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. Adv. Neural. Inf. Process. Syst. 34, 17864\u201317875 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4619_CR35","doi-asserted-by":"crossref","unstructured":"Cheng, B., Misra, I., Schwing, A.G., et al.: Masked-attention mask transformer for universal image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1290\u20131299 (2022)","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"4619_CR36","doi-asserted-by":"publisher","first-page":"8580","DOI":"10.1109\/TMM.2023.3238521","volume":"25","author":"H Liu","year":"2023","unstructured":"Liu, H., Peng, P., Chen, T., et al.: Fecanet: Boosting few-shot semantic segmentation with feature-enhanced context-aware network. IEEE Trans. Multimedia 25, 8580\u20138592 (2023)","journal-title":"IEEE Trans. Multimedia"},{"key":"4619_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, Y., He, N., Yang, J., Li, Y., Wei, D., Huang, Y., Zhang, Y., He, Z., Zheng, Y.: mmformer: Multimodal medical transformer for incomplete multimodal learning of brain tumor segmentation. In: International Conference on Medical Image Computing and Computer-assisted Intervention, pp. 107\u2013117 (2022). Springer","DOI":"10.1007\/978-3-031-16443-9_11"},{"key":"4619_CR38","doi-asserted-by":"crossref","unstructured":"Shaban, A., Bansal, S., Liu, Z., Essa, I., Boots, B.: One-shot learning for semantic segmentation. arXiv:1709.03410 (2017)","DOI":"10.5244\/C.31.167"},{"key":"4619_CR39","doi-asserted-by":"crossref","unstructured":"Saito, K., Saenko, K., Liu, M.-Y.: Coco-funit: Few-shot unsupervised image translation with a content conditioned style encoder. In: European Conference on Computer Vision, pp. 382\u2013398, Springer (2020)","DOI":"10.1007\/978-3-030-58580-8_23"},{"issue":"2","key":"4619_CR40","doi-asserted-by":"publisher","first-page":"1273","DOI":"10.1109\/TPAMI.2023.3329725","volume":"46","author":"X Luo","year":"2024","unstructured":"Luo, X., Tian, Z., Zhang, T., Yu, B., Tang, Y.Y., Jia, J.: Pfenet++: Boosting few-shot semantic segmentation with the noise-filtered context-aware prior mask. IEEE Trans. Pattern Anal. Mach. Intell. 46(2), 1273\u20131289 (2024). https:\/\/doi.org\/10.1109\/TPAMI.2023.3329725","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4619_CR41","doi-asserted-by":"publisher","unstructured":"Ma, J., Xie, G.-S., Zhao, F., Li, Z.: Afanet: Adaptive frequency-aware network for weakly-supervised few-shot semantic segmentation. IEEE Transactions on Multimedia 27, 4018\u20134028 (2025). https:\/\/doi.org\/10.1109\/TMM.2025.3535348","DOI":"10.1109\/TMM.2025.3535348"},{"key":"4619_CR42","doi-asserted-by":"crossref","unstructured":"Xu, D., Yu, S., Zhou, J., Guo, F.: Multi-scale prototype convolutional network for few-shot semantic segmentation. PLOS One 20(4) (2025)","DOI":"10.1371\/journal.pone.0319905"},{"issue":"1","key":"4619_CR43","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1007\/s11263-023-01886-8","volume":"132","author":"C Lang","year":"2024","unstructured":"Lang, C., Cheng, G., Tu, B., et al.: Few-shot segmentation via divide-and-conquer proxies. Int. J. Comput. Vis. 132(1), 261\u2013283 (2024)","journal-title":"Int. J. Comput. Vis."},{"key":"4619_CR44","doi-asserted-by":"publisher","first-page":"113149","DOI":"10.1016\/j.knosys.2025.113149","volume":"285","author":"P Li","year":"2025","unstructured":"Li, P., Liu, F., Jiao, L., et al.: Llm knowledge-driven target prototype learning for few-shot segmentation. Knowl.-Based Syst. 285, 113149 (2025)","journal-title":"Knowl.-Based Syst."},{"key":"4619_CR45","doi-asserted-by":"crossref","unstructured":"Li, J., Shi, K., Xie, G.S., et al.: Label-efficient few-shot semantic segmentation with unsupervised meta-training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 3109\u20133117 (2024)","DOI":"10.1609\/aaai.v38i4.28094"},{"key":"4619_CR46","doi-asserted-by":"crossref","unstructured":"Yan, J., Zhuang, X., Zhao, X., et al.: Camsnet: Few-shot semantic segmentation via class activation map and self-cross attention block. Comput. Mater. Contin. 82(3) (2025)","DOI":"10.32604\/cmc.2025.059709"},{"key":"4619_CR47","doi-asserted-by":"crossref","unstructured":"Shen, J., Kuang, K., Wang, J., et al.: Cgmgm: A cross-gaussian mixture generative model for few-shot semantic segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 4784\u20134792 (2024)","DOI":"10.1609\/aaai.v38i5.28280"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04619-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04619-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04619-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:38:06Z","timestamp":1784893086000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04619-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":47,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["4619"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04619-3","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"13 April 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","label":"Ethics approval and consent to participate","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":3,"name":"Ethics","label":"Consent for publication","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":4,"name":"Ethics","label":"Materials availability","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The source code used in this study is available from the corresponding author upon reasonable request.","order":5,"name":"Ethics","label":"Code availability","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":6,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"386"}}