{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T20:26:42Z","timestamp":1784665602891,"version":"3.55.0"},"reference-count":40,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.imavis.2026.105987","type":"journal-article","created":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T16:28:08Z","timestamp":1776011288000},"page":"105987","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Injecting image text structure and edge priors into segment anything for scene text segmentation"],"prefix":"10.1016","volume":"170","author":[{"given":"Qian","family":"Shao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Libo","family":"Weng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanjing","family":"Lei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanming","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xianxun","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hui","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.105987_b1","series-title":"Artificial Neural Networks and Machine Learning \u2013 ICANN 2019: Image Processing","first-page":"238","article-title":"Coco_ts dataset: Pixel\u2013level annotations based on weak supervision for scene text segmentation","author":"Bonechi","year":"2019"},{"key":"10.1016\/j.imavis.2026.105987_b2","series-title":"2017 14th IAPR International Conference on Document Analysis and Recognition","first-page":"935","article-title":"Total-text: A comprehensive dataset for scene text detection and recognition","volume":"01","author":"Ch\u2019ng","year":"2017"},{"key":"10.1016\/j.imavis.2026.105987_b3","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"2898","article-title":"Scene text segmentation with text-focused transformers","author":"Yu","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b4","series-title":"2013 12th International Conference on Document Analysis and Recognition","first-page":"1484","article-title":"ICDAR 2013 robust reading competition","author":"Karatzas","year":"2013"},{"issue":"2","key":"10.1016\/j.imavis.2026.105987_b5","doi-asserted-by":"crossref","first-page":"225","DOI":"10.1016\/S0031-3203(99)00055-2","article-title":"Adaptive document image binarization","volume":"33","author":"Sauvola","year":"2000","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.imavis.2026.105987_b6","series-title":"Proceedings of the 9th IAPR International Workshop on Document Analysis Systems","first-page":"159","article-title":"Binarization of historical document images using the local maximum and minimum","author":"Su","year":"2010"},{"key":"10.1016\/j.imavis.2026.105987_b7","series-title":"2006 IEEE International Conference on Multimedia and Expo","first-page":"1721","article-title":"Multiscale edge-based text extraction from complex images","author":"Liu","year":"2006"},{"key":"10.1016\/j.imavis.2026.105987_b8","series-title":"2016 IEEE International Conference on Image Processing","first-page":"3763","article-title":"Learning document image binarization from data","author":"Wu","year":"2016"},{"key":"10.1016\/j.imavis.2026.105987_b9","doi-asserted-by":"crossref","unstructured":"H. Zhao, J. Shi, X. Qi, X. Wang, J. Jia, Pyramid Scene Parsing Network, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, CVPR, 2017.","DOI":"10.1109\/CVPR.2017.660"},{"key":"10.1016\/j.imavis.2026.105987_b10","series-title":"2023 IEEE International Conference on Multimedia and Expo","first-page":"1877","article-title":"TextFormer: Component-aware text segmentation with transformer","author":"Wang","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b11","series-title":"Computer Vision \u2013 ECCV 2024","first-page":"410","article-title":"Eaformer: Scene text segmentation with edge-aware transformers","author":"Yu","year":"2025"},{"key":"10.1016\/j.imavis.2026.105987_b12","doi-asserted-by":"crossref","unstructured":"A. Kirillov, E. Mintun, N. Ravi, H. Mao, C. Rolland, L. Gustafson, T. Xiao, S. Whitehead, A.C. Berg, W.-Y. Lo, P. Dollar, R. Girshick, Segment Anything, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, ICCV, 2023, pp. 4015\u20134026.","DOI":"10.1109\/ICCV51070.2023.00371"},{"issue":"3","key":"10.1016\/j.imavis.2026.105987_b13","doi-asserted-by":"crossref","first-page":"1431","DOI":"10.1109\/TPAMI.2024.3495831","article-title":"Hi-sam: Marrying segment anything model for hierarchical text segmentation","volume":"47","author":"Ye","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.imavis.2026.105987_b14","series-title":"International Conference on Document Analysis and Recognition","first-page":"163","article-title":"CHSAM: Efficient scene text segmentation via SAM with convolutional adapters and hierarchical decoding","author":"Zhang","year":"2025"},{"key":"10.1016\/j.imavis.2026.105987_b15","first-page":"29914","article-title":"Segment anything in high quality","volume":"vol. 36","author":"Ke","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b16","doi-asserted-by":"crossref","unstructured":"Z. Xie, B. Guan, W. Jiang, M. Yi, Y. Ding, H. Lu, L. Zhang, PA-SAM: Prompt Adapter SAM for High-Quality Image Segmentation, in: 2024 IEEE International Conference on Multimedia and Expo, ICME, 2024, pp. 1\u20136.","DOI":"10.1109\/ICME57554.2024.10687602"},{"key":"10.1016\/j.imavis.2026.105987_b17","doi-asserted-by":"crossref","unstructured":"X. Xu, Z. Zhang, Z. Wang, B. Price, Z. Wang, H. Shi, Rethinking Text Segmentation: A Novel Dataset and a Text-Specific Refinement Approach, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2021, pp. 12045\u201312055.","DOI":"10.1109\/CVPR46437.2021.01187"},{"key":"10.1016\/j.imavis.2026.105987_b18","series-title":"Personalize segment anything model with one shot","author":"Zhang","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b19","series-title":"AI-SAM: Automatic and interactive segment anything model","author":"Pan","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b20","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.128395","article-title":"Multi-scale contrastive adaptor learning for segmenting anything in underperformed scenes","volume":"606","author":"Zhou","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.imavis.2026.105987_b21","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"2445","article-title":"SAM-DEBLUR: Let segment anything boost image deblurring","author":"Li","year":"2024"},{"key":"10.1016\/j.imavis.2026.105987_b22","series-title":"Medical Image Understanding and Analysis","first-page":"187","article-title":"Adaptivesam: Towards efficient tuning of SAM for surgical scene segmentation","author":"Paranjape","year":"2024"},{"key":"10.1016\/j.imavis.2026.105987_b23","series-title":"Medical SAM adapter: Adapting segment anything model for medical image segmentation","author":"Wu","year":"2023"},{"key":"10.1016\/j.imavis.2026.105987_b24","doi-asserted-by":"crossref","DOI":"10.1016\/j.media.2024.103310","article-title":"MA-SAM: Modality-agnostic SAM adaptation for 3D medical image segmentation","volume":"98","author":"Chen","year":"2024","journal-title":"Med. Image Anal."},{"key":"10.1016\/j.imavis.2026.105987_b25","doi-asserted-by":"crossref","unstructured":"P. Zhang, T. Yan, Y. Liu, H. Lu, Fantastic Animals and Where to Find Them: Segment Any Marine Animal with Dual SAM, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2024, pp. 2578\u20132587.","DOI":"10.1109\/CVPR52733.2024.00249"},{"key":"10.1016\/j.imavis.2026.105987_b26","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.129122","article-title":"Clipsam: CLIP and SAM collaboration for zero-shot anomaly segmentation","volume":"618","author":"Li","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.imavis.2026.105987_b27","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.129420","article-title":"SA3D-l: A lightweight model for 3D object segmentation using neural radiance fields","volume":"623","author":"Liu","year":"2025","journal-title":"Neurocomputing"},{"issue":"7","key":"10.1016\/j.imavis.2026.105987_b28","doi-asserted-by":"crossref","first-page":"1315","DOI":"10.14716\/ijtech.v10i7.3263","article-title":"Hardware-based sobel gradient computations for sharpness enhancement","volume":"10","author":"Kho","year":"2019","journal-title":"Int. J. Technol."},{"key":"10.1016\/j.imavis.2026.105987_b29","series-title":"European Conference on Computer Vision","first-page":"199","article-title":"Adaptive spatial-bce loss for weakly supervised semantic segmentation","author":"Wu","year":"2022"},{"key":"10.1016\/j.imavis.2026.105987_b30","series-title":"2020 IEEE International Conference on Data Mining","first-page":"851","article-title":"Rethinking dice loss for medical image segmentation","author":"Zhao","year":"2020"},{"key":"10.1016\/j.imavis.2026.105987_b31","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.imavis.2026.105987_b32","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.patrec.2020.06.023","article-title":"Weak supervision for generating pixel\u2013level annotations in scene text segmentation","volume":"138","author":"Bonechi","year":"2020","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.imavis.2026.105987_b33","doi-asserted-by":"crossref","unstructured":"X. Xu, Z. Zhang, Z. Wang, B. Price, Z. Wang, H. Shi, Rethinking text segmentation: A novel dataset and a text-specific refinement approach, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 12045\u201312055.","DOI":"10.1109\/CVPR46437.2021.01187"},{"key":"10.1016\/j.imavis.2026.105987_b34","doi-asserted-by":"crossref","unstructured":"L.-C. Chen, Y. Zhu, G. Papandreou, F. Schroff, H. Adam, Encoder-Decoder with Atrous Separable Convolution for Semantic Image Segmentation, in: Proceedings of the European Conference on Computer Vision, ECCV, 2018.","DOI":"10.1007\/978-3-030-01234-2_49"},{"issue":"10","key":"10.1016\/j.imavis.2026.105987_b35","doi-asserted-by":"crossref","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","article-title":"Deep high-resolution representation learning for visual recognition","volume":"43","author":"Wang","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.imavis.2026.105987_b36","series-title":"Computer Vision \u2013 ECCV 2020","first-page":"173","article-title":"Object-contextual representations for semantic segmentation","author":"Yuan","year":"2020"},{"key":"10.1016\/j.imavis.2026.105987_b37","series-title":"Adapting segment anything model for power transmission corridor hazard segmentation","author":"Chen","year":"2025"},{"key":"10.1016\/j.imavis.2026.105987_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.cmpb.2024.108564","article-title":"A client\u2013server based recognition system: Non-contact single\/multiple emotional and behavioral state assessment methods","volume":"260","author":"Zhu","year":"2025","journal-title":"Comput. Methods Programs Biomed."},{"issue":"5","key":"10.1016\/j.imavis.2026.105987_b39","doi-asserted-by":"crossref","first-page":"62","DOI":"10.1109\/MIS.2025.3597120","article-title":"A generative random modality dropout framework for robust multimodal emotion recognition","volume":"40","author":"Zhang","year":"2025","journal-title":"IEEE Intell. Syst."},{"key":"10.1016\/j.imavis.2026.105987_b40","doi-asserted-by":"crossref","DOI":"10.1016\/j.array.2025.100445","article-title":"Raft: robust adversarial fusion transformer for multimodal sentiment analysis","author":"Wang","year":"2025","journal-title":"Array"}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000946?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000946?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T11:18:40Z","timestamp":1777288720000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0262885626000946"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":40,"alternative-id":["S0262885626000946"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105987","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Injecting image text structure and edge priors into segment anything for scene text segmentation","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105987","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"105987"}}