{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:18:06Z","timestamp":1783315086971,"version":"3.54.6"},"reference-count":98,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T00:00:00Z","timestamp":1776384000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T00:00:00Z","timestamp":1776384000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Liaoning Province Science and Technology Plan Joint Program","award":["2024JH2\/102600089"],"award-info":[{"award-number":["2024JH2\/102600089"]}]},{"name":"the Natural Science Foundation of Liaoning Province","award":["2025-BS-0276"],"award-info":[{"award-number":["2025-BS-0276"]}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62506062"],"award-info":[{"award-number":["62506062"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00530-026-02327-5","type":"journal-article","created":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T11:49:20Z","timestamp":1776426560000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["QHSP-Net: query-aware higher-order statistical pooling network for referring image segmentation"],"prefix":"10.1007","volume":"32","author":[{"given":"Qiule","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianxin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bingbing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peihua","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,17]]},"reference":[{"key":"2327_CR1","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, M., Darrell, T.: Segmentation from natural language expressions. In: Proceedings of the European Conference on Computer Vision, pp. 108\u2013124 (2016)","DOI":"10.1007\/978-3-319-46448-0_7"},{"key":"2327_CR2","doi-asserted-by":"crossref","unstructured":"Wang, Z., Lu, Y., Li, Q., Tao, X., Guo, Y., Gong, M., Liu, T.: Cris: clip-driven referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11686\u201311695 (2022)","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"2327_CR3","doi-asserted-by":"crossref","unstructured":"Xu, Z., Chen, Z., Zhang, Y., Song, Y., Wan, X., Li, G.: Bridging vision and language encoders: parameter-efficient tuning for referring image segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17503\u201317512 (2023)","DOI":"10.1109\/ICCV51070.2023.01605"},{"key":"2327_CR4","doi-asserted-by":"crossref","unstructured":"Shang, C., Song, Z., Qiu, H., Wang, L., Meng, F., Li, H.: Prompt-driven referring image segmentation with instance contrasting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4124\u20134134 (2024)","DOI":"10.1109\/CVPR52733.2024.00395"},{"key":"2327_CR5","doi-asserted-by":"crossref","unstructured":"Dai, M., Cheng, W., Liu, J.-j., Yang, S., Cai, W., Sun, Y., Yang, W.: Deris: decoupling perception and cognition for enhanced referring image segmentation through loopback synergy. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2025)","DOI":"10.1109\/ICCV51701.2025.01854"},{"key":"2327_CR6","doi-asserted-by":"crossref","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: Proceedings of the European Conference on Computer Vision, pp. 69\u201385 (2016)","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"2327_CR7","doi-asserted-by":"crossref","unstructured":"Yu, L., Lin, Z., Shen, X., Yang, J., Lu, X., Bansal, M., Berg, T.L.: Mattnet: modular attention network for referring expression comprehension. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1307\u20131315 (2018)","DOI":"10.1109\/CVPR.2018.00142"},{"key":"2327_CR8","doi-asserted-by":"crossref","unstructured":"Huang, P.-H., Lee, H.-H., Chen, H.-T., Liu, T.-L.: Text-guided graph neural networks for referring 3d instance segmentation. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 1610\u20131618 (2021)","DOI":"10.1609\/aaai.v35i2.16253"},{"key":"2327_CR9","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, M., Andreas, J., Darrell, T., Saenko, K.: Modeling relationships in referential expressions with compositional modular networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1115\u20131124 (2017)","DOI":"10.1109\/CVPR.2017.470"},{"key":"2327_CR10","doi-asserted-by":"crossref","unstructured":"Liu, X., Wang, Z., Shao, J., Wang, X., Li, H.: Improving referring expression grounding with cross-modal attention-guided erasing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1950\u20131959 (2019)","DOI":"10.1109\/CVPR.2019.00205"},{"key":"2327_CR11","doi-asserted-by":"publisher","first-page":"3657","DOI":"10.1109\/TMM.2022.3163578","volume":"25","author":"C Liu","year":"2022","unstructured":"Liu, C., Jiang, X., Ding, H.: Instance-specific feature propagation for referring segmentation. IEEE Trans. Multimedia 25, 3657\u20133667 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"2327_CR12","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Jin, P., Li, H., Li, K., Li, S., Ji, X., Liu, C., Chen, J.: Wico: win-win cooperation of bottom-up and top-down referring image segmentation. In: International Joint Conference on Artificial Intelligence, pp. 636\u2013644 (2023)","DOI":"10.24963\/ijcai.2023\/71"},{"key":"2327_CR13","doi-asserted-by":"crossref","unstructured":"Yu, S., Seo, P.H., Son, J.: Zero-shot referring image segmentation with global-local context features. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19456\u201319465 (2023)","DOI":"10.1109\/CVPR52729.2023.01864"},{"key":"2327_CR14","doi-asserted-by":"crossref","unstructured":"Liu, T., Li, S.: Hybrid global-local representation with augmented spatial guidance for zero-shot referring image segmentation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 29634\u201329643 (2025)","DOI":"10.1109\/CVPR52734.2025.02759"},{"key":"2327_CR15","doi-asserted-by":"crossref","unstructured":"Wang, Y., Ni, J., Liu, Y., Yuan, C., Tang, Y.: Iterprime: zero-shot referring image segmentation with iterative grad-cam refinement and primary word emphasis. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 8159\u20138168 (2025)","DOI":"10.1609\/aaai.v39i8.32880"},{"key":"2327_CR16","doi-asserted-by":"crossref","unstructured":"Liu, F., Liu, Y., Kong, Y., Xu, K., Zhang, L., Yin, B., Hancke, G., Lau, R.: Referring image segmentation using text supervision. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 22124\u201322134 (2023)","DOI":"10.1109\/ICCV51070.2023.02022"},{"key":"2327_CR17","doi-asserted-by":"crossref","unstructured":"Kim, D., Kim, N., Lan, C., Kwak, S.: Shatter and gather: learning referring image segmentation with text supervision. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15547\u201315557 (2023)","DOI":"10.1109\/ICCV51070.2023.01425"},{"key":"2327_CR18","doi-asserted-by":"crossref","unstructured":"Dai, Q., Yang, S.: Curriculum point prompting for weakly-supervised referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13711\u201313722 (2024)","DOI":"10.1109\/CVPR52733.2024.01301"},{"key":"2327_CR19","doi-asserted-by":"crossref","unstructured":"Cheng, S., Liu, Y., He, X., Ourselin, S., Tan, L., Luo, G.: Weakmcn: multi-task collaborative network for weakly supervised referring expression comprehension and segmentation. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 9175\u20139185 (2025)","DOI":"10.1109\/CVPR52734.2025.00857"},{"key":"2327_CR20","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"2327_CR21","doi-asserted-by":"publisher","first-page":"34892","DOI":"10.52202\/075280-1516","volume":"36","author":"H Liu","year":"2023","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. Adv. Neural. Inf. Process. Syst. 36, 34892\u201334916 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2327_CR22","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., Dong, L., Bjorck, J., Peng, Z., Liu, Q., Aggarwal, K., Mohammed, O.K., Singhal, S., Som, S., et al.: Image as a foreign language: beit pretraining for vision and vision-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19175\u201319186 (2023)","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"2327_CR23","unstructured":"Bai, J., Bai, S., Yang, S., Wang, S., Tan, S., Wang, P., Lin, J., Zhou, C., Zhou, J.: Qwen-VL: a versatile vision-language model for understanding, localization, text reading, and beyond (2024)"},{"key":"2327_CR24","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J., et al.: Qwen2.5-VL technical report. arXiv preprint arXiv:2502.13923 (2025)"},{"key":"2327_CR25","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., Ge, W., et al.: Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint arXiv:2409.12191 (2024)"},{"key":"2327_CR26","doi-asserted-by":"crossref","unstructured":"Lai, X., Tian, Z., Chen, Y., Li, Y., Yuan, Y., Liu, S., Jia, J.: Lisa: reasoning segmentation via large language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9579\u20139589 (2024)","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"2327_CR27","unstructured":"Liu, Y., Ma, Z., Pu, J., Qi, Z., Wu, Y., Ying, S., Chen, C.W.: Unipixel: unified object referring and segmentation for pixel-level visual reasoning. In: Advances in Neural Information Processing Systems (2025)"},{"key":"2327_CR28","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., RoyChowdhury, A., Maji, S.: Bilinear cnn models for fine-grained visual recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1449\u20131457 (2015)","DOI":"10.1109\/ICCV.2015.170"},{"key":"2327_CR29","doi-asserted-by":"crossref","unstructured":"Li, P., Xie, J., Wang, Q., Gao, Z.: Towards faster training of global covariance pooling networks by iterative matrix square root normalization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 947\u2013955 (2018)","DOI":"10.1109\/CVPR.2018.00105"},{"key":"2327_CR30","doi-asserted-by":"crossref","unstructured":"Chen, B., Deng, W., Hu, J.: Mixed high-order attention network for person re-identification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 371\u2013381 (2019)","DOI":"10.1109\/ICCV.2019.00046"},{"issue":"8","key":"2327_CR31","first-page":"2582","volume":"43","author":"Q Wang","year":"2021","unstructured":"Wang, Q., Xie, J., Zuo, W., Zhang, L., Li, P.: Deep cnns meet global covariance pooling: better representation and generalization. IEEE Trans. Pattern Anal. Mach. Intell. 43(8), 2582\u20132597 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"12","key":"2327_CR32","doi-asserted-by":"publisher","first-page":"15802","DOI":"10.1109\/TPAMI.2023.3321392","volume":"45","author":"Q Wang","year":"2023","unstructured":"Wang, Q., Zhang, Z., Gao, M., Xie, J., Zhu, P., Li, P., Zuo, W., Hu, Q.: Towards a deeper understanding of global covariance pooling in deep learning: an optimization perspective. IEEE Trans. Pattern Anal. Mach. Intell. 45(12), 15802\u201315819 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2327_CR33","unstructured":"Wu, J., Wang, Q., Xie, J., Zhu, P., Hu, Q.: Asymmetric factorized bilinear operation for vision transformer. In: The International Conference on Learning Representations (2025)"},{"key":"2327_CR34","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, J., Tang, Y., Chen, K., Zhao, H., Torr, P.H.: Lavt: language-aware vision transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18155\u201318165 (2022)","DOI":"10.1109\/CVPR52688.2022.01762"},{"key":"2327_CR35","doi-asserted-by":"crossref","unstructured":"Ding, H., Liu, C., Wang, S., Jiang, X.: Vision-language transformer and query generation for referring segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16321\u201316330 (2021)","DOI":"10.1109\/ICCV48922.2021.01601"},{"issue":"6","key":"2327_CR36","doi-asserted-by":"publisher","first-page":"7900","DOI":"10.1109\/TPAMI.2022.3217852","volume":"45","author":"H Ding","year":"2022","unstructured":"Ding, H., Liu, C., Wang, S., Jiang, X.: Vlt: vision-language transformer and query generation for referring segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 45(6), 7900\u20137916 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2327_CR37","doi-asserted-by":"crossref","unstructured":"Shah, N.A., VS, V., Patel, V.M.: Lqmformer: language-aware query mask transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12903\u201312913 (2024)","DOI":"10.1109\/CVPR52733.2024.01226"},{"key":"2327_CR38","doi-asserted-by":"crossref","unstructured":"Chng, Y.X., Zheng, H., Han, Y., Qiu, X., Huang, G.: Mask grounding for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26573\u201326583 (2024)","DOI":"10.1109\/CVPR52733.2024.02509"},{"key":"2327_CR39","doi-asserted-by":"crossref","unstructured":"Tang, J., Zheng, G., Shi, C., Yang, S.: Contrastive grouping with transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23570\u201323580 (2023)","DOI":"10.1109\/CVPR52729.2023.02257"},{"key":"2327_CR40","doi-asserted-by":"publisher","first-page":"1782","DOI":"10.1109\/TIP.2024.3371348","volume":"33","author":"J Wu","year":"2024","unstructured":"Wu, J., Li, X., Li, X., Ding, H., Tong, Y., Tao, D.: Toward robust referring image segmentation. IEEE Trans. Image Process. 33, 1782\u20131794 (2024)","journal-title":"IEEE Trans. Image Process."},{"key":"2327_CR41","doi-asserted-by":"crossref","unstructured":"Huang, J., Xu, Z., Liu, T., Liu, Y., Han, H., Yuan, K., Li, X.: Densely connected parameter-efficient tuning for referring image segmentation. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 3653\u20133661 (2025)","DOI":"10.1609\/aaai.v39i4.32380"},{"key":"2327_CR42","doi-asserted-by":"crossref","unstructured":"Yang, Y., Ma, C., Yao, J., Zhong, Z., Zhang, Y., Wang, Y.: Remamber: referring image segmentation with mamba twister. In: Proceedings of the European Conference on Computer Vision, pp. 108\u2013126 (2024)","DOI":"10.1007\/978-3-031-72684-2_7"},{"issue":"10","key":"2327_CR43","doi-asserted-by":"publisher","first-page":"5999","DOI":"10.1109\/TCSVT.2023.3263468","volume":"33","author":"H Li","year":"2023","unstructured":"Li, H., Sun, M., Xiao, J., Lim, E.G., Zhao, Y.: Fully and weakly supervised referring expression segmentation with end-to-end learning. IEEE Trans. Circuits Syst. Video Technol. 33(10), 5999\u20136012 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2327_CR44","doi-asserted-by":"crossref","unstructured":"Yang, Z., Liu, Y., Lin, J., Hancke, G., Lau, R.: Boosting weakly supervised referring image segmentation via progressive comprehension. In: Advances in Neural Information Processing Systems, pp. 93213\u201393239 (2024)","DOI":"10.52202\/079017-2958"},{"issue":"3","key":"2327_CR45","doi-asserted-by":"publisher","first-page":"3927","DOI":"10.1109\/TNNLS.2022.3201372","volume":"35","author":"G Feng","year":"2022","unstructured":"Feng, G., Zhang, L., Hu, Z., Lu, H.: Learning from box annotations for referring image segmentation. IEEE Transactions on Neural Networks and Learning Systems 35(3), 3927\u20133937 (2022)","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"2327_CR46","doi-asserted-by":"crossref","unstructured":"Qu, M., Wu, Y., Wei, Y., Liu, W., Liang, X., Zhao, Y.: Learning to segment every referring object point by point. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3021\u20133030 (2023)","DOI":"10.1109\/CVPR52729.2023.00295"},{"key":"2327_CR47","unstructured":"Yang, D., Ji, J., Ma, Y., Guo, T., Wang, H., Sun, X., Ji, R.: Sam as the guide: mastering pseudo-label refinement in semi-supervised referring expression segmentation. arXiv preprint arXiv:2406.01451 (2024)"},{"key":"2327_CR48","doi-asserted-by":"crossref","unstructured":"Yu, S., Seo, P.H., Son, J.: Pseudo-ris: distinctive pseudo-supervision generation for referring image segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 18\u201336 (2024)","DOI":"10.1007\/978-3-031-73113-6_2"},{"key":"2327_CR49","doi-asserted-by":"crossref","unstructured":"Nag, S., Goswami, K., Karanam, S.: Safari: adaptive sequence transformer for weakly supervised referring expression segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 485\u2013503 (2024)","DOI":"10.1007\/978-3-031-72784-9_27"},{"key":"2327_CR50","doi-asserted-by":"crossref","unstructured":"Suo, Y., Zhu, L., Yang, Y.: Text augmented spatial-aware zero-shot referring image segmentation. In: Findings of the Association for Computational Linguistics of Empirical Methods in Natural Language Processing, pp. 1032\u20131043 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.73"},{"key":"2327_CR51","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900 (2022)"},{"key":"2327_CR52","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.-Y., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2327_CR53","doi-asserted-by":"crossref","unstructured":"Bao, X., Sun, S., Ma, S., Zheng, K., Guo, Y., Zhao, G., Zheng, Y., Wang, X.: Cores: orchestrating the dance of reasoning and segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 187\u2013204 (2024)","DOI":"10.1007\/978-3-031-72649-1_11"},{"key":"2327_CR54","doi-asserted-by":"crossref","unstructured":"Xia, Z., Han, D., Han, Y., Pan, X., Song, S., Huang, G.: Gsva: generalized segmentation via multimodal large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3858\u20133869 (2024)","DOI":"10.1109\/CVPR52733.2024.00370"},{"key":"2327_CR55","doi-asserted-by":"crossref","unstructured":"Chen, Y.-C., Li, W.-H., Sun, C., Wang, Y.-C.F., Chen, C.-S.: Sam4mllm: enhance multi-modal large language model for referring expression segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 323\u2013340 (2024)","DOI":"10.1007\/978-3-031-73004-7_19"},{"key":"2327_CR56","doi-asserted-by":"crossref","unstructured":"Zhu, L., Chen, T., Xu, Q., Liu, X., Ji, D., Wu, H., Soh, D.W., Liu, J.: Popen: preference-based optimization and ensemble for lvlm-based reasoning segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 30231\u201330240 (2025)","DOI":"10.1109\/CVPR52734.2025.02814"},{"key":"2327_CR57","doi-asserted-by":"crossref","unstructured":"Yu, S., Hong, J., Lee, J., Son, J.: Latent expression generation for referring image segmentation and grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2025)","DOI":"10.1109\/ICCV51701.2025.01985"},{"key":"2327_CR58","doi-asserted-by":"crossref","unstructured":"Dai, M., Li, J., Zhuang, J., Zhang, X., Yang, W.: Multi-task visual grounding with coarse-to-fine consistency constraints. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 2618\u20132626 (2025)","DOI":"10.1609\/aaai.v39i3.32265"},{"issue":"6","key":"2327_CR59","doi-asserted-by":"publisher","first-page":"1309","DOI":"10.1109\/TPAMI.2017.2723400","volume":"40","author":"TY Lin","year":"2018","unstructured":"Lin, T.Y., RoyChowdhury, A., Maji, S.: Bilinear convolutional neural networks for fine-grained visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. 40(6), 1309\u20131322 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2327_CR60","doi-asserted-by":"crossref","unstructured":"Li, P., Xie, J., Wang, Q., Zuo, W.: Is second-order information helpful for large-scale visual recognition? In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2070\u20132078 (2017)","DOI":"10.1109\/ICCV.2017.228"},{"key":"2327_CR61","doi-asserted-by":"crossref","unstructured":"Wang, Q., Zhang, L., Wu, B., Ren, D., Li, P., Zuo, W., Hu, Q.: What deep cnns benefit from global covariance pooling: an optimization perspective. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10771\u201310780 (2020)","DOI":"10.1109\/CVPR42600.2020.01078"},{"key":"2327_CR62","doi-asserted-by":"crossref","unstructured":"Gao, Z., Xie, J., Wang, Q., Li, P.: Global second-order pooling convolutional networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3024\u20133033 (2019)","DOI":"10.1109\/CVPR.2019.00314"},{"key":"2327_CR63","doi-asserted-by":"crossref","unstructured":"Wang, Q., Gao, M., Zhang, Z., Xie, J., Li, P., Hu, Q.: Dropcov: a simple yet effective method for improving deep architectures. In: Advances in Neural Information Processing Systems, pp. 33576\u201333588 (2022)","DOI":"10.52202\/068431-2433"},{"key":"2327_CR64","unstructured":"Oquab, M., Darcet, T., Moutakanni, T., Vo, H.V., Szafraniec, M., Khalidov, V., Fernandez, P., HAZIZA, D., Massa, F., El-Nouby, A., Assran, M., Ballas, N., Galuba, W., Howes, R., Huang, P.-Y., Li, S.-W., Misra, I., Rabbat, M., Sharma, V., Synnaeve, G., Xu, H., Jegou, H., Mairal, J., Labatut, P., Joulin, A., Bojanowski, P.: DINOv2: learning robust visual features without supervision. Transactions on Machine Learning Research (2024)"},{"key":"2327_CR65","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11\u201320 (2016)","DOI":"10.1109\/CVPR.2016.9"},{"key":"2327_CR66","doi-asserted-by":"crossref","unstructured":"Nagaraja, V.K., Morariu, V.I., Davis, L.S.: Modeling context between objects for referring expression understanding. In: Proceedings of the European Conference on Computer Vision, pp. 792\u2013807 (2016)","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"2327_CR67","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: common objects in context. In: Proceedings of the European Conference on Computer Vision, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2327_CR68","unstructured":"Wang, Y., Li, J., Zhang, X., Shi, B., Li, C., Dai, W., Xiong, H., Tian, Q.: Barleria: An efficient tuning framework for referring image segmentation. In: The International Conference on Learning Representations (2024)"},{"key":"2327_CR69","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., Gelly, S.: Parameter-efficient transfer learning for nlp. In: International Conference on Machine Learning, pp. 2790\u20132799 (2019)"},{"key":"2327_CR70","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L., et al.: Pytorch: an imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems, pp. 8024\u20138035 (2019)"},{"key":"2327_CR71","doi-asserted-by":"crossref","unstructured":"Jing, Y., Kong, T., Wang, W., Wang, L., Li, L., Tan, T.: Locate then segment: A strong pipeline for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9858\u20139867 (2021)","DOI":"10.1109\/CVPR46437.2021.00973"},{"key":"2327_CR72","doi-asserted-by":"crossref","unstructured":"Hu, Z., Feng, G., Sun, J., Zhang, L., Lu, H.: Bi-directional relationship inferring network for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4424\u20134433 (2020)","DOI":"10.1109\/CVPR42600.2020.00448"},{"key":"2327_CR73","doi-asserted-by":"crossref","unstructured":"Luo, G., Zhou, Y., Sun, X., Cao, L., Wu, C., Deng, C., Ji, R.: Multi-task collaborative network for joint referring expression comprehension and segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10034\u201310043 (2020)","DOI":"10.1109\/CVPR42600.2020.01005"},{"key":"2327_CR74","doi-asserted-by":"crossref","unstructured":"Hui, T., Liu, S., Huang, S., Li, G., Yu, S., Zhang, F., Han, J.: Linguistic structure guided context modeling for referring image segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 59\u201375 (2020)","DOI":"10.1007\/978-3-030-58607-2_4"},{"key":"2327_CR75","doi-asserted-by":"crossref","unstructured":"Huang, S., Hui, T., Liu, S., Li, G., Wei, Y., Han, J., Liu, L., Li, B.: Referring image segmentation via cross-modal progressive comprehension. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10488\u201310497 (2020)","DOI":"10.1109\/CVPR42600.2020.01050"},{"issue":"9","key":"2327_CR76","first-page":"4761","volume":"44","author":"S Liu","year":"2021","unstructured":"Liu, S., Hui, T., Huang, S., Wei, Y., Li, B., Li, G.: Cross-modal progressive comprehension for referring segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 44(9), 4761\u20134775 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2327_CR77","doi-asserted-by":"crossref","unstructured":"Feng, G., Hu, Z., Zhang, L., Lu, H.: Encoder fusion network with co-attention embedding for referring image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15506\u201315515 (2021)","DOI":"10.1109\/CVPR46437.2021.01525"},{"key":"2327_CR78","doi-asserted-by":"crossref","unstructured":"Kim, N., Kim, D., Lan, C., Zeng, W., Kwak, S.: Restr: Convolution-free referring image segmentation using transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18145\u201318154 (2022)","DOI":"10.1109\/CVPR52688.2022.01761"},{"key":"2327_CR79","doi-asserted-by":"crossref","unstructured":"Hu, Y., Wang, Q., Shao, W., Xie, E., Li, Z., Han, J., Luo, P.: Beyond one-to-one: Rethinking the referring image segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4067\u20134077 (2023)","DOI":"10.1109\/ICCV51070.2023.00376"},{"key":"2327_CR80","doi-asserted-by":"crossref","unstructured":"Ouyang, S., Wang, H., Xie, S., Niu, Z., Tong, R., Chen, Y.-W., Lin, L.: Slvit: scale-wise language-guided vision transformer for referring image segmentation. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, pp. 1294\u20131302 (2023)","DOI":"10.24963\/ijcai.2023\/144"},{"key":"2327_CR81","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, J., Tang, Y., Chen, K., Zhao, H., Torr, P.H.: Semantics-aware dynamic localization and refinement for referring image segmentation. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 3222\u20133230 (2023)","DOI":"10.1609\/aaai.v37i3.25428"},{"key":"2327_CR82","doi-asserted-by":"crossref","unstructured":"Liu, C., Ding, H., Jiang, X.: Gres: generalized referring expression segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23592\u201323601 (2023)","DOI":"10.1109\/CVPR52729.2023.02259"},{"key":"2327_CR83","doi-asserted-by":"crossref","unstructured":"Liu, J., Ding, H., Cai, Z., Zhang, Y., Satzoda, R.K., Mahadevan, V., Manmatha, R.: Polyformer: referring image segmentation as sequential polygon generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18653\u201318663 (2023)","DOI":"10.1109\/CVPR52729.2023.01789"},{"key":"2327_CR84","doi-asserted-by":"crossref","unstructured":"Zhao, W., Rao, Y., Liu, Z., Liu, B., Zhou, J., Lu, J.: Unleashing text-to-image diffusion models for visual perception. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5729\u20135739 (2023)","DOI":"10.1109\/ICCV51070.2023.00527"},{"key":"2327_CR85","doi-asserted-by":"crossref","unstructured":"Kim, S., Kang, M., Kim, D., Park, J., Kwak, S.: Extending clip\u2019s image-text alignment to referring image segmentation. In: North American Chapter of the Association for Computational Linguistics, pp. 4611\u20134628 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.258"},{"key":"2327_CR86","doi-asserted-by":"crossref","unstructured":"Ren, Z., Huang, Z., Wei, Y., Zhao, Y., Fu, D., Feng, J., Jin, X.: Pixellm: Pixel reasoning with large multimodal model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26374\u201326383 (2024)","DOI":"10.1109\/CVPR52733.2024.02491"},{"key":"2327_CR87","doi-asserted-by":"crossref","unstructured":"Pi, R., Yao, L., Gao, J., Zhang, J., Zhang, T.: Perceptiongpt: effectively fusing visual perception into llm. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27124\u201327133 (2024)","DOI":"10.1109\/CVPR52733.2024.02561"},{"key":"2327_CR88","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Ma, Z., Gao, X., Shakiah, S., Gao, Q., Chai, J.: Groundhog: grounding large language models to holistic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14227\u201314238 (2024)","DOI":"10.1109\/CVPR52733.2024.01349"},{"key":"2327_CR89","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Maaz, M., Shaji, S., Shaker, A., Khan, S., Cholakkal, H., Anwer, R.M., Xing, E., Yang, M.-H., Khan, F.S.: Glamm: pixel grounding large multimodal model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13009\u201313018 (2024)","DOI":"10.1109\/CVPR52733.2024.01236"},{"key":"2327_CR90","doi-asserted-by":"crossref","unstructured":"Qian, R., Yin, X., Dou, D.: Reasoning to attend: try to understand how<seg> token works. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 24722\u201324731 (2025)","DOI":"10.1109\/CVPR52734.2025.02302"},{"key":"2327_CR91","unstructured":"Jang, D., Cho, Y., Lee, S., Kim, T., Kim, D.: MMR: a large-scale benchmark dataset for multi-target and multi-granularity reasoning segmentation. In: The International Conference on Learning Representations (2025)"},{"key":"2327_CR92","unstructured":"Wang, X., Zhang, S., Li, S., Li, K., Kallidromitis, K., Kato, Y., Kozuka, K., Darrell, T.: SegLLM: multi-round reasoning segmentation with large language models. In: The International Conference on Learning Representations (2025)"},{"key":"2327_CR93","doi-asserted-by":"crossref","unstructured":"Xiao, L., Yang, X., Peng, F., Wang, Y., Xu, C.: Oneref: Unified one-tower expression grounding and segmentation with mask referring modeling. In: Advances in Neural Information Processing Systems, pp. 139854\u2013139885 (2024)","DOI":"10.52202\/079017-4438"},{"key":"2327_CR94","doi-asserted-by":"crossref","unstructured":"Zheng, L., Chiang, W.-L., Sheng, Y., Zhuang, S., Wu, Z., Zhuang, Y., Lin, Z., Li, Z., Li, D., Xing, E., et al.: Judging llm-as-a-judge with mt-bench and chatbot arena. In: Advances in Neural Information Processing Systems, pp. 46595\u201346623 (2023)","DOI":"10.52202\/075280-2020"},{"key":"2327_CR95","unstructured":"Li, M., Sigal, L.: Referring transformer: A one-step approach to multi-task visual grounding. In: Advances in Neural Information Processing Systems, pp. 19652\u201319664 (2021)"},{"key":"2327_CR96","doi-asserted-by":"crossref","unstructured":"Su, W., Miao, P., Dou, H., Wang, G., Qiao, L., Li, Z., Li, X.: Language adaptive weight generation for multi-task visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10857\u201310866 (2023)","DOI":"10.1109\/CVPR52729.2023.01045"},{"key":"2327_CR97","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Li, K., Jin, P., Li, S., Ji, X., Yuan, L., Liu, C., Chen, J.: Parallel vertex diffusion for unified visual grounding. In: Proceedings of the Association for the Advancement of Artificial Intelligence, pp. 1326\u20131334 (2024)","DOI":"10.1609\/aaai.v38i2.27896"},{"key":"2327_CR98","doi-asserted-by":"crossref","unstructured":"Sun, S., Li, R., Torr, P., Gu, X., Li, S.: Clip as rnn: Segment countless visual concepts without training endeavor. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13171\u201313182 (2024)","DOI":"10.1109\/CVPR52733.2024.01251"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02327-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02327-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02327-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:06:12Z","timestamp":1783314372000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02327-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,17]]},"references-count":98,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2327"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02327-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,17]]},"assertion":[{"value":"22 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"All authors have approved the manuscript and agree with its publication.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and Informed Consent for Data Used"}}],"article-number":"241"}}