{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:57:24Z","timestamp":1783439844827,"version":"3.54.6"},"publisher-location":"Cham","reference-count":77,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031198144","type":"print"},{"value":"9783031198151","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-19815-1_17","type":"book-chapter","created":{"date-parts":[[2022,10,19]],"date-time":"2022-10-19T23:11:54Z","timestamp":1666221114000},"page":"284-302","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":35,"title":["Language Matters: A Weakly Supervised Vision-Language Pre-training Approach for Scene Text Detection and Spotting"],"prefix":"10.1007","author":[{"given":"Chuhui","family":"Xue","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenqing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shijian","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Philip H. S.","family":"Torr","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Song","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,10,20]]},"reference":[{"key":"17_CR1","doi-asserted-by":"crossref","unstructured":"Baek, J., Matsui, Y., Aizawa, K.: What if we only use real datasets for scene text recognition? toward scene text recognition with fewer labels. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3113\u20133122 (2021)","DOI":"10.1109\/CVPR46437.2021.00313"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Baek, Y., Lee, B., Han, D., Yun, S., Lee, H.: Character region awareness for text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9365\u20139374 (2019)","DOI":"10.1109\/CVPR.2019.00959"},{"key":"17_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1007\/978-3-030-58526-6_30","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y Baek","year":"2020","unstructured":"Baek, Y., et al.: Character region attention for text spotting. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12374, pp. 504\u2013521. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58526-6_30"},{"key":"17_CR4","doi-asserted-by":"crossref","unstructured":"Chen, Y.C., Li, L., Yu, L., El Kholy, A., Ahmed, F., Gan, Z., Cheng, Y., Liu, J.: Uniter: learning universal image-text representations (2019)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"17_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"YC Chen","year":"2020","unstructured":"Chen, Y.C., et al.: UNITER: uNiversal image-TExt representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 104\u2013120. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"key":"17_CR6","doi-asserted-by":"publisher","first-page":"50441","DOI":"10.1109\/ACCESS.2021.3069041","volume":"9","author":"MJ Chiou","year":"2021","unstructured":"Chiou, M.J., Zimmermann, R., Feng, J.: Visual relationship detection with visual-linguistic knowledge from multimodal representations. IEEE Access 9, 50441\u201350451 (2021)","journal-title":"IEEE Access"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Ch\u2019ng, C.K., Chan, C.S.: Total-text: a comprehensive dataset for scene text detection and recognition. In: 2017 14th IAPR International Conference on Document Analysis and Recognition (ICDAR), vol. 1, pp. 935\u2013942. IEEE (2017)","DOI":"10.1109\/ICDAR.2017.157"},{"key":"17_CR8","doi-asserted-by":"crossref","unstructured":"Dai, P., Zhang, S., Zhang, H., Cao, X.: Progressive contour regression for arbitrary-shape scene text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7393\u20137402 (2021)","DOI":"10.1109\/CVPR46437.2021.00731"},{"key":"17_CR9","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"17_CR10","doi-asserted-by":"crossref","unstructured":"Feng, W., He, W., Yin, F., Zhang, X.Y., Liu, C.L.: Textdragon: an end-to-end framework for arbitrary shaped text spotting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9076\u20139085 (2019)","DOI":"10.1109\/ICCV.2019.00917"},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Gupta, A., Vedaldi, A., Zisserman, A.: Synthetic data for text localisation in natural images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2315\u20132324 (2016)","DOI":"10.1109\/CVPR.2016.254"},{"key":"17_CR12","doi-asserted-by":"crossref","unstructured":"Hao, W., Li, C., Li, X., Carin, L., Gao, J.: Towards learning a generic agent for vision-and-language navigation via pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13137\u201313146 (2020)","DOI":"10.1109\/CVPR42600.2020.01315"},{"key":"17_CR13","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"17_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"17_CR15","doi-asserted-by":"crossref","unstructured":"He, M., et al.: Most: a multi-oriented scene text detector with localization refinement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8813\u20138822 (2021)","DOI":"10.1109\/CVPR46437.2021.00870"},{"key":"17_CR16","doi-asserted-by":"crossref","unstructured":"He, T., Tian, Z., Huang, W., Shen, C., Qiao, Y., Sun, C.: An end-to-end textspotter with explicit alignment and attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5020\u20135029 (2018)","DOI":"10.1109\/CVPR.2018.00527"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"Karatzas, D., et al.: ICDAR 2015 competition on robust reading. In: 2015 13th International Conference on Document Analysis and Recognition (ICDAR), pp. 1156\u20131160. IEEE (2015)","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"17_CR18","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"17_CR19","doi-asserted-by":"crossref","unstructured":"Kittenplon, Y., Lavi, I., Fogel, S., Bar, Y., Manmatha, R., Perona, P.: Towards weakly-supervised text spotting using a multi-task transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4604\u20134613 (2022)","DOI":"10.1109\/CVPR52688.2022.00456"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Li, G., Duan, N., Fang, Y., Gong, M., Jiang, D.: Unicoder-vl: a universal encoder for vision and language by cross-modal pre-training. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, no. 07, pp. 11336\u201311344 (2020)","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Li, H., Wang, P., Shen, C.: Towards end-to-end text spotting with convolutional recurrent neural networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5238\u20135246 (2017)","DOI":"10.1109\/ICCV.2017.560"},{"key":"17_CR22","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: Visualbert: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)"},{"issue":"2","key":"17_CR23","doi-asserted-by":"publisher","first-page":"532","DOI":"10.1109\/TPAMI.2019.2937086","volume":"43","author":"M Liao","year":"2021","unstructured":"Liao, M., Lyu, P., He, M., Yao, C., Wu, W., Bai, X.: Mask textspotter: an end-to-end trainable neural network for spotting text with arbitrary shapes. IEEE Trans. Pattern Anal. Mach. Intell. 43(2), 532\u2013548 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2019.2937086","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"17_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"706","DOI":"10.1007\/978-3-030-58621-8_41","volume-title":"Computer Vision \u2013 ECCV 2020","author":"M Liao","year":"2020","unstructured":"Liao, M., Pang, G., Huang, J., Hassner, T., Bai, X.: Mask TextSpotter v3: segmentation proposal network for robust scene text spotting. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12356, pp. 706\u2013722. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58621-8_41"},{"issue":"8","key":"17_CR25","doi-asserted-by":"publisher","first-page":"3676","DOI":"10.1109\/TIP.2018.2825107","volume":"27","author":"M Liao","year":"2018","unstructured":"Liao, M., Shi, B., Bai, X.: Textboxes++: a single-shot oriented scene text detector. IEEE Trans. Image Process. 27(8), 3676\u20133690 (2018)","journal-title":"IEEE Trans. Image Process."},{"issue":"2","key":"17_CR26","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11432-019-2737-0","volume":"63","author":"M Liao","year":"2020","unstructured":"Liao, M., Song, B., Long, S., He, M., Yao, C., Bai, X.: Synthtext3d: synthesizing scene text images from 3D virtual worlds. Sci. China Inf. Sci. 63(2), 1\u201314 (2020)","journal-title":"Sci. China Inf. Sci."},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Liao, M., Wan, Z., Yao, C., Chen, K., Bai, X.: Real-time scene text detection with differentiable binarization. In: Proceedings of AAAI, vol. 34, no. 07, pp. 11474\u201311481 (2020)","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Liao, M., Zhu, Z., Shi, B., Xia, G.s., Bai, X.: Rotation-sensitive regression for oriented scene text detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5909\u20135918 (2018)","DOI":"10.1109\/CVPR.2018.00619"},{"key":"17_CR29","doi-asserted-by":"crossref","unstructured":"Liu, X., Liang, D., Yan, S., Chen, D., Qiao, Y., Yan, J.: Fots: fast oriented text spotting with a unified network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5676\u20135685 (2018)","DOI":"10.1109\/CVPR.2018.00595"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Liu, Y., Chen, H., Shen, C., He, T., Jin, L., Wang, L.: Abcnet: real-time scene text spotting with adaptive bezier-curve network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9809\u20139818 (2020)","DOI":"10.1109\/CVPR42600.2020.00983"},{"key":"17_CR31","doi-asserted-by":"publisher","unstructured":"Liu, Y., et al.: Abcnet v2: adaptive bezier-curve network for real-time end-to-end text spotting. IEEE Trans. Pattern Anal. Mach. Intell. 1 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2021.3107437","DOI":"10.1109\/TPAMI.2021.3107437"},{"key":"17_CR32","doi-asserted-by":"crossref","unstructured":"Long, S., Ruan, J., Zhang, W., He, X., Wu, W., Yao, C.: Textsnake: a flexible representation for detecting text of arbitrary shapes. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 20\u201336 (2018)","DOI":"10.1007\/978-3-030-01216-8_2"},{"key":"17_CR33","unstructured":"Loshchilov, I., Hutter, F.: Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)"},{"key":"17_CR34","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"17_CR35","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"17_CR36","doi-asserted-by":"crossref","unstructured":"Lyu, P., Liao, M., Yao, C., Wu, W., Bai, X.: Mask textspotter: an end-to-end trainable neural network for spotting text with arbitrary shapes. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 67\u201383 (2018)","DOI":"10.1007\/978-3-030-01264-9_5"},{"key":"17_CR37","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1007\/978-3-030-58539-6_16","volume-title":"Computer Vision \u2013 ECCV 2020","author":"A Majumdar","year":"2020","unstructured":"Majumdar, A., Shrivastava, A., Lee, S., Anderson, P., Parikh, D., Batra, D.: Improving vision-and-language navigation with image-text Pairs from the web. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12351, pp. 259\u2013274. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58539-6_16"},{"key":"17_CR38","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"336","DOI":"10.1007\/978-3-030-58523-5_20","volume-title":"Computer Vision \u2013 ECCV 2020","author":"V Murahari","year":"2020","unstructured":"Murahari, V., Batra, D., Parikh, D., Das, A.: Large-scale pretraining for visual dialog: a simple state-of-the-art baseline. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12363, pp. 336\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58523-5_20"},{"key":"17_CR39","doi-asserted-by":"crossref","unstructured":"Qiao, L., et al.: Mango: a mask attention guided one-stage scene text spotter. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 2467\u20132476 (2021)","DOI":"10.1609\/aaai.v35i3.16348"},{"key":"17_CR40","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"17_CR41","doi-asserted-by":"crossref","unstructured":"Shi, B., Bai, X., Belongie, S.: Detecting oriented text in natural images by linking segments. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2550\u20132558 (2017)","DOI":"10.1109\/CVPR.2017.371"},{"key":"17_CR42","doi-asserted-by":"crossref","unstructured":"Shi, B., Bai, X., Belongie, S.: Detecting oriented text in natural images by linking segments. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2550\u20132558 (2017)","DOI":"10.1109\/CVPR.2017.371"},{"key":"17_CR43","unstructured":"Su, W., et al.: Vl-bert: pre-training of generic visual-linguistic representations. arXiv preprint arXiv:1908.08530 (2019)"},{"key":"17_CR44","doi-asserted-by":"crossref","unstructured":"Sun, Y., Liu, J., Liu, W., Han, J., Ding, E., Liu, J.: Chinese street view text: Large-scale Chinese text reading with partially supervised learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9086\u20139095 (2019)","DOI":"10.1109\/ICCV.2019.00918"},{"key":"17_CR45","doi-asserted-by":"crossref","unstructured":"Sun, Y., et al.: ICDAR 2019 competition on large-scale street view text with partial labeling-RRC-LSVT. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1557\u20131562. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00250"},{"key":"17_CR46","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"17_CR47","doi-asserted-by":"crossref","unstructured":"Tang, J., Yang, Z., Wang, Y., Zheng, Q., Xu, Y., Bai, X.: Seglink++: detecting dense and arbitrary-shaped scene text by instance-aware component grouping. Pattern Recogn. 96, 106954 (2019)","DOI":"10.1016\/j.patcog.2019.06.020"},{"key":"17_CR48","doi-asserted-by":"crossref","unstructured":"Tensmeyer, C., Wigington, C.: Training full-page handwritten text recognition models without annotated line breaks. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1\u20138. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00011"},{"key":"17_CR49","doi-asserted-by":"crossref","unstructured":"Tian, S., Lu, S., Li, C.: Wetext: scene text detection under weak supervision. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1492\u20131500 (2017)","DOI":"10.1109\/ICCV.2017.166"},{"key":"17_CR50","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"17_CR51","doi-asserted-by":"crossref","unstructured":"Wan, Q., Ji, H., Shen, L.: Self-attention based text knowledge mining for text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5983\u20135992 (2021)","DOI":"10.1109\/CVPR46437.2021.00592"},{"key":"17_CR52","doi-asserted-by":"crossref","unstructured":"Wang, F., Zhao, L., Li, X., Wang, X., Tao, D.: Geometry-aware scene text detection with instance transformation network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1381\u20131389 (2018)","DOI":"10.1109\/CVPR.2018.00150"},{"key":"17_CR53","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: All you need is boundary: toward arbitrary-shaped text spotting. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 12160\u201312167 (2020)","DOI":"10.1609\/aaai.v34i07.6896"},{"key":"17_CR54","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Hou, W., Lu, T., Yu, G., Shao, S.: Shape robust text detection with progressive scale expansion network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9336\u20139345 (2019)","DOI":"10.1109\/CVPR.2019.00956"},{"key":"17_CR55","doi-asserted-by":"crossref","unstructured":"Wang, W., et alC.: Pan++: towards efficient and accurate end-to-end spotting of arbitrarily-shaped text. IEEE Trans. Pattern Anal. Mach. Intell. 44(9), 5349\u20135367 (2021)","DOI":"10.1109\/TPAMI.2021.3077555"},{"key":"17_CR56","doi-asserted-by":"crossref","unstructured":"Wang, W., et al.: Efficient and accurate arbitrary-shaped text detection with pixel aggregation network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8440\u20138449 (2019)","DOI":"10.1109\/ICCV.2019.00853"},{"key":"17_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., Jiang, Y., Luo, Z., Liu, C.L., Choi, H., Kim, S.: Arbitrary shape scene text detection with adaptive text region representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6449\u20136458 (2019)","DOI":"10.1109\/CVPR.2019.00661"},{"key":"17_CR58","doi-asserted-by":"crossref","unstructured":"Wang, Y., Joty, S., Lyu, M.R., King, I., Xiong, C., Hoi, S.C.: Vd-bert: a unified vision and dialog transformer with bert. arXiv preprint arXiv:2004.13278 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.269"},{"key":"17_CR59","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xie, H., Zha, Z.J., Xing, M., Fu, Z., Zhang, Y.: Contournet: taking a further step toward accurate arbitrary-shaped scene text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11753\u201311762 (2020)","DOI":"10.1109\/CVPR42600.2020.01177"},{"key":"17_CR60","doi-asserted-by":"crossref","unstructured":"Wu, W., et al.: Synthetic-to-real unsupervised domain adaptation for scene text detection in the wild. In: Proceedings of the Asian Conference on Computer Vision (2020)","DOI":"10.1007\/978-3-030-69535-4_18"},{"key":"17_CR61","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1007\/978-3-030-58526-6_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Xiao","year":"2020","unstructured":"Xiao, S., Peng, L., Yan, R., An, K., Yao, G., Min, J.: Sequential deformation for accurate scene text detection. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12374, pp. 108\u2013124. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58526-6_7"},{"key":"17_CR62","doi-asserted-by":"crossref","unstructured":"Xing, L., Tian, Z., Huang, W., Scott, M.R.: Convolutional character networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9126\u20139136 (2019)","DOI":"10.1109\/ICCV.2019.00922"},{"issue":"11","key":"17_CR63","doi-asserted-by":"publisher","first-page":"5566","DOI":"10.1109\/TIP.2019.2900589","volume":"28","author":"Y Xu","year":"2019","unstructured":"Xu, Y., Wang, Y., Zhou, W., Wang, Y., Yang, Z., Bai, X.: Textfield: learning a deep direction field for irregular scene text detection. IEEE Trans. Image Process. 28(11), 5566\u20135579 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"17_CR64","unstructured":"Xue, C., Lu, S., Bai, S., Zhang, W., Wang, C.: I2c2w: image-to-character-to-word transformers for accurate scene text recognition. arXiv preprint arXiv:2105.08383 (2021)"},{"key":"17_CR65","doi-asserted-by":"crossref","unstructured":"Xue, C., Lu, S., Hoi, S.: Detection and rectification of arbitrary shaped scene texts by using text keypoints and links. Pattern Recogn. 124, 108494 (2022)","DOI":"10.1016\/j.patcog.2021.108494"},{"key":"17_CR66","doi-asserted-by":"crossref","unstructured":"Xue, C., Lu, S., Zhan, F.: Accurate scene text detection through border semantics awareness and bootstrapping. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 355\u2013372 (2018)","DOI":"10.1007\/978-3-030-01270-0_22"},{"key":"17_CR67","doi-asserted-by":"crossref","unstructured":"Xue, C., Lu, S., Zhang, W.: Msr: Multi-scale shape regression for scene text detection. arXiv preprint arXiv:1901.02596 (2019)","DOI":"10.24963\/ijcai.2019\/139"},{"key":"17_CR68","unstructured":"Xue, H., et al.: Probing inter-modality: visual parsing with self-attention for vision-and-language pre-training. Adv. Neural Inf. Process. Syst. 34, 4514\u20134528 (2021)"},{"key":"17_CR69","doi-asserted-by":"crossref","unstructured":"Yu, D., et al.: Towards accurate scene text recognition with semantic reasoning networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12113\u201312122 (2020)","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"17_CR70","unstructured":"Yuliang, L., Lianwen, J., Shuaitao, Z., Sheng, Z.: Detecting curve text in the wild: New dataset and new solution. arXiv preprint arXiv:1712.02170 (2017)"},{"key":"17_CR71","doi-asserted-by":"crossref","unstructured":"Zhan, F., Lu, S., Xue, C.: Verisimilar image synthesis for accurate detection and recognition of texts in scenes. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 249\u2013266 (2018)","DOI":"10.1007\/978-3-030-01237-3_16"},{"key":"17_CR72","doi-asserted-by":"crossref","unstructured":"Zhan, F., Xue, C., Lu, S.: Ga-dan: Geometry-aware domain adaptation network for scene text detection and recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9105\u20139115 (2019)","DOI":"10.1109\/ICCV.2019.00920"},{"key":"17_CR73","doi-asserted-by":"crossref","unstructured":"Zhang, C., Liang, B., Huang, Z., En, M., Han, J., Ding, E., Ding, X.: Look more than once: an accurate detector for text of arbitrary shapes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10552\u201310561 (2019)","DOI":"10.1109\/CVPR.2019.01080"},{"key":"17_CR74","doi-asserted-by":"crossref","unstructured":"Zhang, S.X., et al.: Deep relational reasoning graph network for arbitrary shape text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9699\u20139708 (2020)","DOI":"10.1109\/CVPR42600.2020.00972"},{"key":"17_CR75","doi-asserted-by":"crossref","unstructured":"Zhang, S.X., Zhu, X., Yang, C., Wang, H., Yin, X.C.: Adaptive boundary proposal network for arbitrary shape text detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1305\u20131314 (2021)","DOI":"10.1109\/ICCV48922.2021.00134"},{"key":"17_CR76","doi-asserted-by":"crossref","unstructured":"Zhou, X., et al.: East: an efficient and accurate scene text detector. In: The IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5551\u20135560 (2017)","DOI":"10.1109\/CVPR.2017.283"},{"key":"17_CR77","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Chen, J., Liang, L., Kuang, Z., Jin, L., Zhang, W.: Fourier contour embedding for arbitrary-shaped text detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3123\u20133131 (2021)","DOI":"10.1109\/CVPR46437.2021.00314"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-19815-1_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T04:35:50Z","timestamp":1728189350000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-19815-1_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031198144","9783031198151"],"references-count":77,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-19815-1_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"20 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}