{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T05:18:23Z","timestamp":1769059103221,"version":"3.49.0"},"reference-count":41,"publisher":"MDPI AG","issue":"12","license":[{"start":{"date-parts":[[2022,12,1]],"date-time":"2022-12-01T00:00:00Z","timestamp":1669852800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172267"],"award-info":[{"award-number":["62172267"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61936001"],"award-info":[{"award-number":["61936001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key R&amp;D Program of China","award":["62172267"],"award-info":[{"award-number":["62172267"]}]},{"name":"National Key R&amp;D Program of China","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}]},{"name":"National Key R&amp;D Program of China","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}]},{"name":"National Key R&amp;D Program of China","award":["61936001"],"award-info":[{"award-number":["61936001"]}]},{"name":"National Key R&amp;D Program of China","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}]},{"name":"National Key R&amp;D Program of China","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["62172267"],"award-info":[{"award-number":["62172267"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["61936001"],"award-info":[{"award-number":["61936001"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}]},{"name":"Natural Science Foundation of Shanghai, China","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["62172267"],"award-info":[{"award-number":["62172267"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["61936001"],"award-info":[{"award-number":["61936001"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}]},{"name":"State Key Program of National Natural Science Foundation of China","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}]},{"name":"Shanghai Pujiang Program","award":["62172267"],"award-info":[{"award-number":["62172267"]}]},{"name":"Shanghai Pujiang Program","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}]},{"name":"Shanghai Pujiang Program","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}]},{"name":"Shanghai Pujiang Program","award":["61936001"],"award-info":[{"award-number":["61936001"]}]},{"name":"Shanghai Pujiang Program","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}]},{"name":"Shanghai Pujiang Program","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["62172267"],"award-info":[{"award-number":["62172267"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["2019YFE0190500"],"award-info":[{"award-number":["2019YFE0190500"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["20ZR1420400"],"award-info":[{"award-number":["20ZR1420400"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["61936001"],"award-info":[{"award-number":["61936001"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["21PJ1404200"],"award-info":[{"award-number":["21PJ1404200"]}]},{"name":"Key Research Project of Zhejiang Laboratory","award":["2021PE0AC02"],"award-info":[{"award-number":["2021PE0AC02"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Information"],"abstract":"<jats:p>Scene Text Detection (STD) is critical for obtaining textual information from natural scenes, serving for automated driving and security surveillance. However, existing text detection methods fall short when dealing with the variation in text curvatures, orientations, and aspect ratios in complex backgrounds. To meet the challenge, we propose a method called CA-STD to detect arbitrarily shaped text against a complicated background. Firstly, a Feature Refinement Module (FRM) is proposed to enhance feature representation. Additionally, the conditional attention mechanism is proposed not only to decouple the spatial and textual information from scene text images, but also to model the relationship among different feature vectors. Finally, the Contour Information Aggregation (CIA) is presented to enrich the feature representation of text contours by considering circular topology and semantic information simultaneously to obtain the detection curves with arbitrary shapes. The proposed CA-STD method is evaluated on different datasets with extensive experiments. On the one hand, the CA-STD outperforms state-of-the-art methods and achieves 82.9 in precision on the dataset of TotalText. On the other hand, the method has better performance than state-of-the-art methods and achieves the F1 score of 83.8 on the dataset of CTW-1500. The quantitative and qualitative analysis proves that the CA-STD can detect variably shaped scene text effectively.<\/jats:p>","DOI":"10.3390\/info13120565","type":"journal-article","created":{"date-parts":[[2022,12,1]],"date-time":"2022-12-01T04:28:57Z","timestamp":1669868937000},"page":"565","update-policy":"https:\/\/doi.org\/10.3390\/mdpi_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["CA-STD: Scene Text Detection in Arbitrary Shape Based on Conditional Attention"],"prefix":"10.3390","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5331-022X","authenticated-orcid":false,"given":"Xing","family":"Wu","sequence":"first","affiliation":[{"name":"School of Computer Engineering & Science, Shanghai University, Shanghai 200444, China"},{"name":"Zhejiang Laboratory, Hangzhou 311100, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yangyang","family":"Qi","sequence":"additional","affiliation":[{"name":"School of Computer Engineering & Science, Shanghai University, Shanghai 200444, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4479-700X","authenticated-orcid":false,"given":"Jun","family":"Song","sequence":"additional","affiliation":[{"name":"Department of Geography, Faculty of Social Sciences, Hong Kong Baptist University, Hong Kong 999077, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junfeng","family":"Yao","sequence":"additional","affiliation":[{"name":"Cssc Seago System Technology Co., Ltd., Shanghai 200010, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanzhong","family":"Wang","sequence":"additional","affiliation":[{"name":"Shanghai Jianke Engineering Project Management Co., Ltd., Shanghai 200032, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Engineering & Science, Shanghai University, Shanghai 200444, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuexing","family":"Han","sequence":"additional","affiliation":[{"name":"School of Computer Engineering & Science, Shanghai University, Shanghai 200444, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quan","family":"Qian","sequence":"additional","affiliation":[{"name":"School of Computer Engineering & Science, Shanghai University, Shanghai 200444, China"},{"name":"Zhejiang Laboratory, Hangzhou 311100, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"1968","published-online":{"date-parts":[[2022,12,1]]},"reference":[{"key":"ref_1","doi-asserted-by":"crossref","unstructured":"Raisi, Z., Naiel, M.A., and Younes, G. (2021, January 20\u201325). Transformer-based text detection in the wild. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Nashville, TN, USA.","DOI":"10.1109\/CVPRW53098.2021.00353"},{"key":"ref_2","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhang, C., and Shen, W. (2016, January 27\u201330). Multi-oriented text detection with fully convolutional networks. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA.","DOI":"10.1109\/CVPR.2016.451"},{"key":"ref_3","doi-asserted-by":"crossref","first-page":"761","DOI":"10.1016\/j.imavis.2004.02.006","article-title":"Robust wide-baseline stereo from maximally stable extremal regions","volume":"22","author":"Matas","year":"2004","journal-title":"Image Vis. Comput."},{"key":"ref_4","doi-asserted-by":"crossref","first-page":"385","DOI":"10.1016\/j.ins.2022.02.006","article-title":"FTAP: Feature transferring autonomous machine learning pipeline","volume":"593","author":"Wu","year":"2022","journal-title":"Inf. Sci."},{"key":"ref_5","doi-asserted-by":"crossref","first-page":"14665","DOI":"10.1007\/s10489-022-03541-0","article-title":"Face aging with pixel-level alignment GAN","volume":"52","author":"Wu","year":"2022","journal-title":"Appl. Intell."},{"key":"ref_6","doi-asserted-by":"crossref","first-page":"128837","DOI":"10.1109\/ACCESS.2019.2939201","article-title":"A survey of deep learning-based object detection","volume":"7","author":"Jiao","year":"2019","journal-title":"IEEE Access"},{"key":"ref_7","doi-asserted-by":"crossref","unstructured":"Shi, B., Bai, X., and Belongie, S. (2017, January 21\u201326). Detecting oriented text in natural images by linking segments. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, HI, USA.","DOI":"10.1109\/CVPR.2017.371"},{"key":"ref_8","doi-asserted-by":"crossref","unstructured":"Lyu, P., Liao, M., and Yao, C. (2018, January 8\u201314). Mask textspotter: An end-to-end trainable neural network for spotting text with arbitrary shapes. Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany.","DOI":"10.1007\/978-3-030-01264-9_5"},{"key":"ref_9","doi-asserted-by":"crossref","unstructured":"Deng, D., Liu, H., and Li, X. (2018, January 2\u20137). Pixellink: Detecting scene text via instance segmentation. Proceedings of the AAAI Conference on Artificial Intelligence, New Orleans, LA, USA.","DOI":"10.1609\/aaai.v32i1.12269"},{"key":"ref_10","unstructured":"Wang, W., Xie, E., and Song, X. (December, January 27). Efficient and accurate arbitrary-shaped text detection with pixel aggregation network. Proceedings of the IEEE\/CVF International Conference on Computer Vision, Seoul, Korea."},{"key":"ref_11","doi-asserted-by":"crossref","unstructured":"Long, S., Ruan, J., and Zhang, W. (2018, January 8\u201314). Textsnake: A flexible representation for detecting text of arbitrary shapes. Proceedings of the European Conference on Computer Vision (ECCV), Munich, Germany.","DOI":"10.1007\/978-3-030-01216-8_2"},{"key":"ref_12","doi-asserted-by":"crossref","unstructured":"Ye, J., Chen, Z., and Liu, J. (2020, January 12\u201318). TextFuseNet: Scene Text Detection with Richer Fused Features. Proceedings of the IJCAI, Rhodes, Greece.","DOI":"10.24963\/ijcai.2020\/72"},{"key":"ref_13","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1016\/j.ins.2019.08.059","article-title":"The assessment of small bowel motility with attentive deformable neural network","volume":"508","author":"Wu","year":"2020","journal-title":"Inf. Sci."},{"key":"ref_14","doi-asserted-by":"crossref","unstructured":"Wu, X., Jin, H., and Ye, X. (2020). Multiscale convolutional and recurrent neural network for quality prediction of continuous casting slabs. Processes, 9.","DOI":"10.3390\/pr9010033"},{"key":"ref_15","doi-asserted-by":"crossref","unstructured":"Ibrayim, M., Li, Y., and Hamdulla, A. (2022). Scene Text Detection Based on Two-Branch Feature Extraction. Sensors, 22.","DOI":"10.3390\/s22166262"},{"key":"ref_16","doi-asserted-by":"crossref","unstructured":"Hassan, E. (2022). Scene Text Detection Using Attention with Depthwise Separable Convolutions. Appl. Sci., 12.","DOI":"10.3390\/app12136425"},{"key":"ref_17","doi-asserted-by":"crossref","unstructured":"Li, Y., Ibrayim, M., and Hamdulla, A. (2021). CSFF-Net: Scene Text Detection Based on Cross-Scale Feature Fusion. Information, 12.","DOI":"10.3390\/info12120524"},{"key":"ref_18","doi-asserted-by":"crossref","unstructured":"Lyu, P., Yao, C., and Wu, W. (2018, January 18\u201323). Multi-oriented scene text detection via corner localization and region segmentation. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Salt Lake City, UT, USA.","DOI":"10.1109\/CVPR.2018.00788"},{"key":"ref_19","doi-asserted-by":"crossref","unstructured":"Wang, X., Jiang, Y., and Luo, Z. (2019, January 15\u201320). Arbitrary shape scene text detection with adaptive text region representation. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, CA, USA.","DOI":"10.1109\/CVPR.2019.00661"},{"key":"ref_20","doi-asserted-by":"crossref","unstructured":"Liao, M., Zhu, Z., and Shi, B. (2018, January 18\u201323). Rotation-sensitive regression for oriented scene text detection. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Salt Lake City, UT, USA.","DOI":"10.1109\/CVPR.2018.00619"},{"key":"ref_21","doi-asserted-by":"crossref","unstructured":"Liao, M., Shi, B., and Bai, X. (2017, January 4\u20139). Textboxes: A fast text detector with a single deep neural network. Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, San Francisco, CA, USA.","DOI":"10.1609\/aaai.v31i1.11196"},{"key":"ref_22","doi-asserted-by":"crossref","first-page":"3111","DOI":"10.1109\/TMM.2018.2818020","article-title":"Arbitrary-oriented scene text detection via rotation proposals","volume":"20","author":"Ma","year":"2018","journal-title":"IEEE Trans. Multimed."},{"key":"ref_23","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","volume":"39","author":"Ren","year":"2017","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref_24","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xie, H., and Zha, Z.J. (2020, January 13\u201319). Contournet: Taking a further step toward accurate arbitrary-shaped scene text detection. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, WA, USA.","DOI":"10.1109\/CVPR42600.2020.01177"},{"key":"ref_25","doi-asserted-by":"crossref","first-page":"1961","DOI":"10.1007\/s11263-022-01616-6","article-title":"I3CL: Intra-and Inter-Instance Collaborative Learning for Arbitrary-shaped Scene Text Detection","volume":"130","author":"Du","year":"2022","journal-title":"Int. J. Comput. Vis."},{"key":"ref_26","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., and Cao, Y. (2021, January 11\u201317). Swin transformer: Hierarchical vision transformer using shifted windows. Proceedings of the IEEE\/CVF International Conference on Computer Vision, Virtual Event.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref_27","unstructured":"Carion, N., Massa, F., and Synnaeve, G. End-to-end object detection with transformers. Proceedings of the European Conference on Computer Vision."},{"key":"ref_28","unstructured":"Chen, M., Radford, A., and Child, R. (2020, January 13\u201318). Generative pretraining from pixels. Proceedings of the International Conference on Machine Learning, PMLR, Virtual Event."},{"key":"ref_29","doi-asserted-by":"crossref","unstructured":"Liu, R., Yuan, Z., and Liu, T. (2021, January 5\u20139). End-to-end lane shape prediction with transformers. Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, Virtual.","DOI":"10.1109\/WACV48630.2021.00374"},{"key":"ref_30","doi-asserted-by":"crossref","unstructured":"Peng, S., Jiang, W., and Pi, H. (2020, January 13\u201319). Deep snake for real-time instance segmentation. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, WA, USA.","DOI":"10.1109\/CVPR42600.2020.00856"},{"key":"ref_31","doi-asserted-by":"crossref","unstructured":"Wu, X., Qi, Y., and Tang, B. (2021, January 17\u201319). DA-STD: Deformable Attention-Based Scene Text Detection in Arbitrary Shape. Proceedings of the 2021 IEEE International Conference on Progress in Informatics and Computing (PIC), Shanghai, China.","DOI":"10.1109\/PIC53636.2021.9687065"},{"key":"ref_32","doi-asserted-by":"crossref","unstructured":"Gupta, A., Vedaldi, A., and Zisserman, A. (2016, January 27\u201330). Synthetic data for text localisation in natural images. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA.","DOI":"10.1109\/CVPR.2016.254"},{"key":"ref_33","first-page":"935","article-title":"Total-text: A comprehensive dataset for scene text detection and recognition","volume":"Volume 1","author":"Chan","year":"2017","journal-title":"Proceedings of the 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR)"},{"key":"ref_34","doi-asserted-by":"crossref","unstructured":"Baek, Y., Lee, B., and Han, D. (2019, January 15\u201319). Character region awareness for text detection. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, CA, USA.","DOI":"10.1109\/CVPR.2019.00959"},{"key":"ref_35","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref_36","doi-asserted-by":"crossref","unstructured":"Zhang, C., Liang, B., and Huang, Z. (2019, January 15\u201320). Look more than once: An accurate detector for text of arbitrary shapes. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, CA, USA.","DOI":"10.1109\/CVPR.2019.01080"},{"key":"ref_37","doi-asserted-by":"crossref","unstructured":"Wang, P., Zhang, C., and Qi, F. (2019, January 21\u201325). A single-shot arbitrarily-shaped text detector based on context attended multi-task learning. Proceedings of the 27th ACM International Conference on Multimedia, Nice, France.","DOI":"10.1145\/3343031.3350988"},{"key":"ref_38","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Xie, H., and Fang, S. (2020, January 12). CRNet: A center-aware representation for detecting text of arbitrary shapes. Proceedings of the 28th ACM International Conference on Multimedia, Seattle, WA, USA.","DOI":"10.1145\/3394171.3413565"},{"key":"ref_39","unstructured":"Tian, Z., Huang, W., and He, T. Detecting text in natural image with connectionist text proposal network. Proceedings of the European Conference on Computer Vision."},{"key":"ref_40","doi-asserted-by":"crossref","unstructured":"Lin, Z., Zhu, F., and Wang, Q. (2022). RSSGG-CS: Remote Sensing Image Scene Graph Generation by Fusing Contextual Information and Statistical Knowledge. Remote Sens., 14.","DOI":"10.3390\/rs14133118"},{"key":"ref_41","doi-asserted-by":"crossref","unstructured":"Wang, Y., Mamat, H., and Xu, X. (2022). Scene Uyghur Text Detection Based on Fine-Grained Feature Representation. Sensors, 22.","DOI":"10.3390\/s22124372"}],"container-title":["Information"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.mdpi.com\/2078-2489\/13\/12\/565\/pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,11]],"date-time":"2025-10-11T01:31:53Z","timestamp":1760146313000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.mdpi.com\/2078-2489\/13\/12\/565"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,1]]},"references-count":41,"journal-issue":{"issue":"12","published-online":{"date-parts":[[2022,12]]}},"alternative-id":["info13120565"],"URL":"https:\/\/doi.org\/10.3390\/info13120565","relation":{},"ISSN":["2078-2489"],"issn-type":[{"value":"2078-2489","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,1]]}}}