{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:59:48Z","timestamp":1780973988309,"version":"3.54.1"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62006041"],"award-info":[{"award-number":["62006041"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100008081","name":"Southeast University","doi-asserted-by":"publisher","award":["MP202404"],"award-info":[{"award-number":["MP202404"]}],"id":[{"id":"10.13039\/501100008081","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132029","type":"journal-article","created":{"date-parts":[[2026,3,18]],"date-time":"2026-03-18T10:14:35Z","timestamp":1773828875000},"page":"132029","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Cross-modal semantic token alignment via contrastive learning for weakly-supervised referring image segmentation"],"prefix":"10.1016","volume":"319","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9880-8121","authenticated-orcid":false,"given":"Congwei","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1748-8586","authenticated-orcid":false,"given":"Zhibin","family":"Quan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6385-6776","authenticated-orcid":false,"given":"Wankou","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132029_bib0001","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4081","article-title":"RobustSAM: segment anything robustly on degraded images","author":"Chen","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111216","article-title":"Multi-consistency for semi-supervised medical image segmentation via diffusion models","volume":"161","author":"Chen","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132029_bib0003","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"1635","article-title":"Boxsup: Exploiting bounding boxes to supervise convolutional networks for semantic segmentation","author":"Dai","year":"2015"},{"key":"10.1016\/j.eswa.2026.132029_bib0004","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13711","article-title":"Curriculum point prompting for weakly-supervised referring image segmentation","author":"Dai","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0005","series-title":"European conference on computer vision","first-page":"326","article-title":"Segment, select, correct: A framework for weakly-supervised referring segmentation","author":"Eiras","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111204","article-title":"Um-cam: Uncertainty-weighted multi-resolution class activation maps for weakly-supervised segmentation","volume":"160","author":"Fu","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132029_bib0007","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123156","article-title":"An indeterminacy fusion of encoder-decoder network based on neutrosophic set for white blood cells segmentation","volume":"246","author":"Guo","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132029_bib0008","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13332","article-title":"Decoupling static and hierarchical motion perception for referring video segmentation","author":"He","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0009","series-title":"Computer vision\u2013ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11\u201314, 2016, proceedings, Part I 14","first-page":"108","article-title":"Segmentation from natural language expressions","author":"Hu","year":"2016"},{"issue":"10","key":"10.1016\/j.eswa.2026.132029_bib0010","doi-asserted-by":"crossref","first-page":"2962","DOI":"10.3390\/s20102962","article-title":"NextMed: automatic imaging segmentation, 3D reconstruction, and 3D model visualization platform using augmented and virtual reality","volume":"20","author":"Izard","year":"2020","journal-title":"Sensors"},{"key":"10.1016\/j.eswa.2026.132029_bib0011","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127599","article-title":"A survey of methods for addressing the challenges of referring image segmentation","volume":"583","author":"Ji","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.132029_bib0012","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15547","article-title":"Shatter and gather: Learning referring image segmentation with text supervision","author":"Kim","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0013","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4015","article-title":"Segment anything","author":"Kirillov","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19499","article-title":"From sam to cams: Exploring segment anything model for weakly supervised semantic segmentation","author":"Kweon","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0015","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"21870","article-title":"Weakly supervised referring image segmentation with intra-chunk and inter-chunk consistency","author":"Lee","year":"2023"},{"issue":"10","key":"10.1016\/j.eswa.2026.132029_bib0016","doi-asserted-by":"crossref","first-page":"5999","DOI":"10.1109\/TCSVT.2023.3263468","article-title":"Fully and weakly supervised referring expression segmentation with end-to-end learning","volume":"33","author":"Li","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132029_bib0017","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"22124","article-title":"Referring image segmentation using text supervision","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129698","article-title":"PRNet: A progressive refinement network for referring image segmentation","volume":"630","author":"Liu","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.132029_bib0019","doi-asserted-by":"crossref","first-page":"4910","DOI":"10.1109\/TCSVT.2024.3524543","article-title":"ReferSAM: Unleashing segment anything model for referring image segmentation","volume":"35","author":"Liu","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132029_bib0020","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"11","article-title":"Generation and comprehension of unambiguous object descriptions","author":"Mao","year":"2016"},{"key":"10.1016\/j.eswa.2026.132029_bib0021","doi-asserted-by":"crossref","DOI":"10.1016\/j.media.2023.102918","article-title":"Segment anything model for medical image analysis: an experimental study","volume":"89","author":"Mazurowski","year":"2023","journal-title":"Medical Image Analysis"},{"key":"10.1016\/j.eswa.2026.132029_bib0022","unstructured":"McEver, R. A., & Manjunath, B. S. (2020). Pcams: Weakly supervised semantic segmentation using point supervision. arXiv preprint arXiv: 2007.05615."},{"key":"10.1016\/j.eswa.2026.132029_bib0023","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132029_bib0024","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12903","article-title":"LQMFormer: Language-aware query mask transformer for referring image segmentation","author":"Shah","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0025","doi-asserted-by":"crossref","first-page":"28222","DOI":"10.52202\/068431-2046","article-title":"What is where by looking: Weakly-supervised open-world phrase-grounding without text inputs","volume":"35","author":"Shaharabany","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132029_bib0026","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4124","article-title":"Prompt-driven referring image segmentation with instance contrasting","author":"Shang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0027","series-title":"Proceedings of the european conference on computer vision (ECCV)","first-page":"38","article-title":"Key-word-aware network for referring expression image segmentation","author":"Shi","year":"2018"},{"key":"10.1016\/j.eswa.2026.132029_bib0028","unstructured":"Strudel, R., Laptev, I., & Schmid, C. (2022). Weakly-supervised segmentation of referring expressions. arXiv preprint arXiv: 2205.04725."},{"key":"10.1016\/j.eswa.2026.132029_bib0029","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"23570","article-title":"Contrastive grouping with transformer for referring image segmentation","author":"Tang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0030","series-title":"IJCAI international joint conference on artificial intelligence","article-title":"Boundaryperception guidance: A scribble-supervised semantic segmentation approach","author":"Wang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132029_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19175","article-title":"Image as a foreign language: Beit pretraining for vision and vision-language tasks","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0032","article-title":"Ucgm: Enhancing pseudo labels via uncertainty and cross-image gaussian mixture model for semi-supervised semantic segmentation","volume":"296","author":"Wang","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132029_bib0033","unstructured":"Wei, J., & Zou, K. (2019). EDA: Easy data augmentation techniques for boosting performance on text classification tasks. arXiv preprint arXiv: 1901.11196."},{"key":"10.1016\/j.eswa.2026.132029_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112109","article-title":"Towards multimodal sarcasm detection via label-aware graph contrastive learning with back-translation augmentation","volume":"300","author":"Wei","year":"2024","journal-title":"Knowledge-Based Systems"},{"key":"10.1016\/j.eswa.2026.132029_bib0035","unstructured":"Yang, D., Ji, J., Ma, Y., Guo, T., Wang, H., Sun, X., & Ji, R. (2024a). Sam as the guide: mastering pseudo-label refinement in semi-supervised referring expression segmentation. arXiv preprint arXiv: 2406.01451."},{"key":"10.1016\/j.eswa.2026.132029_bib0036","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3606","article-title":"Separate and conquer: Decoupling co-occurrence via decomposition and representation for weakly supervised semantic segmentation","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0037","unstructured":"Yang, Z., Liu, Y., Lin, J., Hancke, G., & Lau, R. W. H. (2024c). Boosting weakly-supervised referring image segmentation via progressive comprehension. arXiv preprint arXiv: 2410.01544."},{"key":"10.1016\/j.eswa.2026.132029_bib0038","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18155","article-title":"Lavt: Language-aware vision transformer for referring image segmentation","author":"Yang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132029_bib0039","series-title":"Computer vision\u2013ECCV 2016: 14th european conference, Amsterdam, the Netherlands, October 11-14, 2016, proceedings, Part II 14","first-page":"69","article-title":"Modeling context in referring expressions","author":"Yu","year":"2016"},{"key":"10.1016\/j.eswa.2026.132029_bib0040","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19456","article-title":"Zero-shot referring image segmentation with global-local context features","author":"Yu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132029_bib0041","doi-asserted-by":"crossref","unstructured":"Yuan, R., Chen, M., Xu, J., Zhou, L., Li, Q., Zhang, Y., Feng, R., Zhang, T., & Gao, S. (2025). Text-promptable propagation for referring medical image sequence segmentation. arXiv preprint arXiv: 2502.11093.","DOI":"10.1145\/3746027.3755166"},{"key":"10.1016\/j.eswa.2026.132029_bib0042","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3796","article-title":"Frozen clip: A strong backbone for weakly supervised semantic segmentation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132029_bib0043","unstructured":"Zhang, C., Liu, L., Cui, Y., Huang, G., Lin, W., Yang, Y., & Hu, Y. (2023). A comprehensive survey on segment anything model for vision and beyond. arXiv preprint arXiv: 2305.08196."},{"key":"10.1016\/j.eswa.2026.132029_bib0044","doi-asserted-by":"crossref","DOI":"10.1016\/j.compbiomed.2024.108238","article-title":"Segment anything model for medical image segmentation: Current applications and future directions","volume":"171","author":"Zhang","year":"2024","journal-title":"Computers in Biology and Medicine"},{"key":"10.1016\/j.eswa.2026.132029_bib0045","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"7525","article-title":"SFC: Shared feature calibration in weakly supervised semantic segmentation","volume":"vol. 38","author":"Zhao","year":"2024"},{"issue":"3","key":"10.1016\/j.eswa.2026.132029_bib0046","doi-asserted-by":"crossref","first-page":"1085","DOI":"10.1007\/s11263-024-02224-2","article-title":"WeakClip: Adapting clip for weakly-supervised semantic segmentation","volume":"133","author":"Zhu","year":"2025","journal-title":"International Journal of Computer Vision"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009425?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009425?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:36:03Z","timestamp":1780972563000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426009425"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":46,"alternative-id":["S0957417426009425"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132029","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Cross-modal semantic token alignment via contrastive learning for weakly-supervised referring image segmentation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132029","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132029"}}