{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T15:10:12Z","timestamp":1758985812156,"version":"3.44.0"},"reference-count":63,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2025,8,1]],"date-time":"2025-08-01T00:00:00Z","timestamp":1754006400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,1]],"date-time":"2025-08-01T00:00:00Z","timestamp":1754006400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61971086,61972062"],"award-info":[{"award-number":["61971086,61972062"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Applied Basic Research Project of Liaoning Province","award":["2023JH2\/101300191,2023JH2\/101300193"],"award-info":[{"award-number":["2023JH2\/101300191,2023JH2\/101300193"]}]},{"name":"Science and Technology Development Program Project of Jilin Province","award":["20230201111GX"],"award-info":[{"award-number":["20230201111GX"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s10489-025-06851-1","type":"journal-article","created":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T16:00:34Z","timestamp":1756569634000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Instance-aware context with mutually guided vision-language attention for referring image segmentation"],"prefix":"10.1007","volume":"55","author":[{"given":"Qiule","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bingbing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7229-3867","authenticated-orcid":false,"given":"Peihua","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,30]]},"reference":[{"key":"6851_CR1","doi-asserted-by":"crossref","unstructured":"Hu R, Rohrbach M, Darrell T (2016) Segmentation from natural language expressions. In: Proceedings of the European conference on computer vision, pp 108\u2013124","DOI":"10.1007\/978-3-319-46448-0_7"},{"key":"6851_CR2","doi-asserted-by":"crossref","unstructured":"Wang Z, Lu Y, Li Q, Tao X, Guo Y, Gong, M Liu T (2022) Cris: Clip-driven referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11686\u201311695","DOI":"10.1109\/CVPR52688.2022.01139"},{"key":"6851_CR3","doi-asserted-by":"crossref","unstructured":"Ye L, Rochan M, Liu Z, Wang Y (2019) Cross-modal self-attention network for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10502\u201310511","DOI":"10.1109\/CVPR.2019.01075"},{"key":"6851_CR4","doi-asserted-by":"crossref","unstructured":"Chen D-J, Jia S, Lo Y-C, Chen H-T, Liu T-L (2019) See-through-text grouping for referring image segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7454\u20137463","DOI":"10.1109\/ICCV.2019.00755"},{"key":"6851_CR5","doi-asserted-by":"crossref","unstructured":"Long J, Shelhamer E, Darrell T (2015) Fully convolutional networks for semantic segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3431\u20133440","DOI":"10.1109\/CVPR.2015.7298965"},{"issue":"4","key":"6851_CR6","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1109\/TPAMI.2016.2572683","volume":"39","author":"E Shelhamer","year":"2017","unstructured":"Shelhamer E, Long J, Darrell T (2017) Fully convolutional networks for semantic segmentation. IEEE Trans Pattern Anal Mach Intell 39(4):640\u2013651","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6851_CR7","doi-asserted-by":"crossref","unstructured":"Yang S, Xia M, Li G, Zhou H-Y, Yu Y (2021) Bottom-up shift and reasoning for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11266\u201311275","DOI":"10.1109\/CVPR46437.2021.01111"},{"key":"6851_CR8","doi-asserted-by":"crossref","unstructured":"Feng G, Hu Z, Zhang L, Lu H (2021) Encoder fusion network with co-attention embedding for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 15506\u201315515","DOI":"10.1109\/CVPR46437.2021.01525"},{"key":"6851_CR9","doi-asserted-by":"crossref","unstructured":"Jing Y, Kong T, Wang W, Wang L, Li L, Tan T (2021) Locate then segment: A strong pipeline for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9858\u20139867","DOI":"10.1109\/CVPR46437.2021.00973"},{"key":"6851_CR10","doi-asserted-by":"crossref","unstructured":"Ding H, Liu C, Wang S, Jiang X (2021) Vision-language transformer and query generation for referring segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 16321\u201316330","DOI":"10.1109\/ICCV48922.2021.01601"},{"key":"6851_CR11","doi-asserted-by":"crossref","unstructured":"Huang S, Hui T, Liu S, Li G, Wei Y, Han J, Liu L, Li B (2020) Referring image segmentation via cross-modal progressive comprehension. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10488\u201310497","DOI":"10.1109\/CVPR42600.2020.01050"},{"issue":"7","key":"6851_CR12","first-page":"3719","volume":"44","author":"L Ye","year":"2021","unstructured":"Ye L, Rochan M, Liu Z, Zhang X, Wang Y (2021) Referring segmentation in images and videos with cross-modal self-attention network. IEEE Trans Pattern Anal Mach Intell 44(7):3719\u20133732","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6851_CR13","doi-asserted-by":"crossref","unstructured":"Yang Z, Wang J, Tang Y, Chen K, Zhao H, Torr PH (2023) Semantics-aware dynamic localization and refinement for referring image segmentation. In: Proceedings of the association for the advancement of artificial intelligence, pp 3222\u20133230","DOI":"10.1609\/aaai.v37i3.25428"},{"key":"6851_CR14","doi-asserted-by":"crossref","unstructured":"Tang J, Zheng G, Shi C, Yang S (2023) Contrastive grouping with transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23570\u201323580","DOI":"10.1109\/CVPR52729.2023.02257"},{"key":"6851_CR15","doi-asserted-by":"crossref","unstructured":"Liu C, Ding H, Jiang X (2023) Gres: Generalized referring expression segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23592\u201323601","DOI":"10.1109\/CVPR52729.2023.02259"},{"key":"6851_CR16","doi-asserted-by":"crossref","unstructured":"Yu S, Seo PH, Son J (2023) Zero-shot referring image segmentation with global-local context features. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19456\u201319465","DOI":"10.1109\/CVPR52729.2023.01864"},{"key":"6851_CR17","doi-asserted-by":"crossref","unstructured":"Yang Z, Wang J, Tang Y, Chen K, Zhao H, Torr PH (2022) Lavt: Language-aware vision transformer for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18155\u201318165","DOI":"10.1109\/CVPR52688.2022.01762"},{"key":"6851_CR18","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et al. (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763"},{"key":"6851_CR19","doi-asserted-by":"crossref","unstructured":"Luo G, Zhou Y, Sun X, Cao L, Wu C, Deng C, Ji R (2020) Multi-task collaborative network for joint referring expression comprehension and segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10034\u201310043","DOI":"10.1109\/CVPR42600.2020.01005"},{"issue":"4","key":"6851_CR20","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"L-C Chen","year":"2017","unstructured":"Chen L-C, Papandreou G, Kokkinos I, Murphy K, Yuille AL (2017) Deeplab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE Trans Pattern Anal Mach Intell 40(4):834\u2013848","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6851_CR21","unstructured":"Chen L-C, Papandreou G, Schroff F, Adam H (2017) Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587."},{"key":"6851_CR22","doi-asserted-by":"crossref","unstructured":"Zhao H, Shi J, Qi X, Wang X, Jia J (2017) Pyramid scene parsing network. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2881\u20132890","DOI":"10.1109\/CVPR.2017.660"},{"key":"6851_CR23","doi-asserted-by":"crossref","unstructured":"Ronneberger O, Fischer P, Brox T (2015) U-net: Convolutional networks for biomedical image segmentation. In: Medical image computing and computer-assisted intervention, pp 234\u2013241","DOI":"10.1007\/978-3-319-24574-4_28"},{"issue":"12","key":"6851_CR24","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan V, Kendall A, Cipolla R (2017) Segnet: A deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans Pattern Anal Mach Intell 39(12):2481\u20132495","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6851_CR25","doi-asserted-by":"crossref","unstructured":"Fu J, Liu J, Tian H, Li Y, Bao Y, Fang Z, Lu H (2019) Dual attention network for scene segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3146\u20133154","DOI":"10.1109\/CVPR.2019.00326"},{"issue":"6","key":"6851_CR26","doi-asserted-by":"publisher","first-page":"2547","DOI":"10.1109\/TNNLS.2020.3006524","volume":"32","author":"J Fu","year":"2020","unstructured":"Fu J, Liu J, Jiang J, Li Y, Bao Y, Lu H (2020) Scene segmentation with dual relation-aware attention network. IEEE Trans Neural Netw Learn Syst 32(6):2547\u20132560","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"6851_CR27","doi-asserted-by":"crossref","unstructured":"Huang Z, Wang X, Huang L, Huang C, Wei Y, Liu W (2019) Ccnet: Criss-cross attention for semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 603\u2013612","DOI":"10.1109\/ICCV.2019.00069"},{"key":"6851_CR28","doi-asserted-by":"crossref","unstructured":"Zhu Z, Xu M, Bai S, Huang T, Bai X (2019) Asymmetric non-local neural networks for semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 593\u2013602","DOI":"10.1109\/ICCV.2019.00068"},{"key":"6851_CR29","doi-asserted-by":"crossref","unstructured":"Zhang H, Dana K, Shi J, Zhang Z, Wang X, Tyagi A, Agrawal A (2018) Context encoding for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 7151\u20137160","DOI":"10.1109\/CVPR.2018.00747"},{"key":"6851_CR30","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1016\/j.neucom.2021.03.003","volume":"445","author":"Q Sun","year":"2021","unstructured":"Sun Q, Zhang Z, Li P (2021) Second-order encoding networks for semantic segmentation. Neurocomputing 445:50\u201360","journal-title":"Neurocomputing"},{"key":"6851_CR31","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"6851_CR32","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, Uszkoreit J, Houlsby N (2021) An image is worth 16x16 words: Transformers for image recognition at scale. Int Conf Learn Represent"},{"key":"6851_CR33","doi-asserted-by":"crossref","unstructured":"Zheng S, Lu J, Zhao H, Zhu X, Luo Z, Wang Y, Fu Y, Feng J, Xiang T, Torr PH et al (2021) Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6881\u20136890","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"6851_CR34","doi-asserted-by":"crossref","unstructured":"Strudel R, Garcia R, Laptev I, Schmid C (2021) Segmenter: Transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7262\u20137272","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"6851_CR35","unstructured":"Xie E, Wang W, Yu Z, Anandkumar A, Alvarez JM, Luo P (2021) Segformer: Simple and efficient design for semantic segmentation with transformers. In: Advances in neural information processing systems, pp 12077\u201312090"},{"key":"6851_CR36","doi-asserted-by":"crossref","unstructured":"Yi M, Cui Q, Wu H, Yang C, Yoshie O, Lu H (2023) A simple framework for text-supervised semantic segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7071\u20137080","DOI":"10.1109\/CVPR52729.2023.00683"},{"key":"6851_CR37","doi-asserted-by":"crossref","unstructured":"Zhang W, Huang Z, Luo G, Chen T, Wang X, Liu W, Yu G, Shen C (2022) Topformer: Token pyramid transformer for mobile semantic segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12083\u201312093","DOI":"10.1109\/CVPR52688.2022.01177"},{"key":"6851_CR38","unstructured":"Kenton JDM-WC, Toutanova LK (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of naacL-HLT, pp 4171\u20134186"},{"key":"6851_CR39","unstructured":"Redmon J, Farhadi A (2018) Yolov3: An incremental improvement. arXiv preprint arXiv:1804.02767"},{"key":"6851_CR40","doi-asserted-by":"crossref","unstructured":"Kim N, Kim D, Lan C, Zeng W, Kwak S (2022) Restr: Convolution-free referring image segmentation using transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18145\u201318154","DOI":"10.1109\/CVPR52688.2022.01761"},{"key":"6851_CR41","doi-asserted-by":"crossref","unstructured":"Xu Z, Chen Z, Zhang Y, Song Y, Wan X, Li G (2023) Bridging vision and language encoders: Parameter-efficient tuning for referring image segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 17503\u201317512","DOI":"10.1109\/ICCV51070.2023.01605"},{"key":"6851_CR42","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Advances in neural information processing systems, pp 5998\u20136008"},{"key":"6851_CR43","doi-asserted-by":"crossref","unstructured":"Shi H, Li H, Meng F, Wu Q (2018) Key-word-aware network for referring expression image segmentation. In: Proceedings of the European Conference on Computer Vision, pp 38\u201354","DOI":"10.1007\/978-3-030-01231-1_3"},{"key":"6851_CR44","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"6851_CR45","doi-asserted-by":"crossref","unstructured":"Kim D, Kim N, Lan C, Kwak S (2023) Shatter and gather: Learning referring image segmentation with text supervision. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 15547\u201315557","DOI":"10.1109\/ICCV51070.2023.01425"},{"key":"6851_CR46","doi-asserted-by":"crossref","unstructured":"Lee J, Lee S, Nam J, Yu S, Do J, Taghavi T (2023) Weakly supervised referring image segmentation with intra-chunk and inter-chunk consistency. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 21870\u201321881","DOI":"10.1109\/ICCV51070.2023.01999"},{"key":"6851_CR47","doi-asserted-by":"crossref","unstructured":"Wang X, Girshick R, Gupta A, He K (2018) Non-local neural networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7794\u20137803","DOI":"10.1109\/CVPR.2018.00813"},{"key":"6851_CR48","doi-asserted-by":"crossref","unstructured":"Yu L, Poirson P, Yang S, Berg AC, Berg TL (2016) Modeling context in referring expressions. In: Proceedings of the European conference on computer vision, pp 69\u201385","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"6851_CR49","doi-asserted-by":"crossref","unstructured":"Mao J, Huang J, Toshev A, Camburu O, Yuille AL, Murphy K (2016) Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11\u201320","DOI":"10.1109\/CVPR.2016.9"},{"key":"6851_CR50","doi-asserted-by":"crossref","unstructured":"Nagaraja VK, Morariu VI, Davis LS (2016) Modeling context between objects for referring expression understanding. In: Proceedings of the European conference on computer vision, pp 792\u2013807","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"6851_CR51","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft coco: Common objects in context. In: Proceedings of the European conference on computer vision, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"6851_CR52","doi-asserted-by":"crossref","unstructured":"Hu Y, Wang Q, Shao W, Xie E, Li Z, Han J, Luo P (2023) Beyond one-to-one: Rethinking the referring image segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4067\u20134077","DOI":"10.1109\/ICCV51070.2023.00376"},{"key":"6851_CR53","unstructured":"Paszke A, Gross S, Massa F, Lerer A, Bradbury J, Chanan G, Killeen T, Lin Z, Gimelshein N, Antiga L et al (2019) Pytorch: An imperative style, high-performance deep learning library. In: Advances in neural information processing systems, pp 8024\u20138035"},{"key":"6851_CR54","doi-asserted-by":"crossref","unstructured":"Liu C, Lin Z, Shen X, Yang J, Lu X, Yuille A (2017) Recurrent multimodal interaction for referring image segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1271\u20131280","DOI":"10.1109\/ICCV.2017.143"},{"key":"6851_CR55","doi-asserted-by":"crossref","unstructured":"Margffoy-Tuay E, P\u00e9rez JC, Botero E, Arbel\u00e1ez P (2018) Dynamic multimodal instance segmentation guided by natural language queries. In: Proceedings of the European conference on computer vision, pp 630\u2013645","DOI":"10.1007\/978-3-030-01252-6_39"},{"key":"6851_CR56","doi-asserted-by":"crossref","unstructured":"Li R, Li K, Kuo Y-C, Shu M, Qi X, Shen X, Jia J (2018) Referring image segmentation via recurrent refinement networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5745\u20135753","DOI":"10.1109\/CVPR.2018.00602"},{"key":"6851_CR57","doi-asserted-by":"crossref","unstructured":"Yu L, Lin Z, Shen X, Yang J, Lu X, Bansal M, Berg TL (2018) Mattnet: Modular attention network for referring expression comprehension. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1307\u20131315","DOI":"10.1109\/CVPR.2018.00142"},{"key":"6851_CR58","doi-asserted-by":"crossref","unstructured":"Liu D, Zhang H, Wu F, Zha Z-J (2019) Learning to assemble neural module tree networks for visual grounding. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4673\u20134682","DOI":"10.1109\/ICCV.2019.00477"},{"key":"6851_CR59","unstructured":"Chen Y-W, Tsai Y-H, Wang T, Lin Y-Y, Yang M-H (2019) Referring expression object segmentation with caption-aware consistency. In: Proceedings of the British machine vision association"},{"key":"6851_CR60","doi-asserted-by":"crossref","unstructured":"Hu Z, Feng G, Sun J, Zhang L, Lu H (2020) Bi-directional relationship inferring network for referring image segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4424\u20134433","DOI":"10.1109\/CVPR42600.2020.00448"},{"key":"6851_CR61","doi-asserted-by":"crossref","unstructured":"Hui T, Liu S, Huang S, Li G, Yu S, Zhang F, Han J (2020) Linguistic structure guided context modeling for referring image segmentation. In: Proceedings of the European conference on computer vision. Springer, pp 59\u201375","DOI":"10.1007\/978-3-030-58607-2_4"},{"key":"6851_CR62","doi-asserted-by":"crossref","unstructured":"Liu F, Liu Y, Kong Y, Xu K, Zhang L, Yin B, Hancke G, Lau R (2023) Referring image segmentation using text supervision. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 22124\u201322134","DOI":"10.1109\/ICCV51070.2023.02022"},{"key":"6851_CR63","doi-asserted-by":"crossref","unstructured":"Suo Y, Zhu L, Yang Y (2023) Text augmented spatial-aware zero-shot referring image segmentation. In: Findings of the association for computational linguistics of empirical methods in natural language processing, pp 1032\u20131043","DOI":"10.18653\/v1\/2023.findings-emnlp.73"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06851-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06851-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06851-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T14:34:32Z","timestamp":1758983672000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06851-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8]]},"references-count":63,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["6851"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06851-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2025,8]]},"assertion":[{"value":"28 August 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 August 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"All authors have approved the manuscript and agree with its publication.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and Informed Consent for Data Used"}}],"article-number":"950"}}