{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T02:53:01Z","timestamp":1781491981690,"version":"3.54.1"},"publisher-location":"Cham","reference-count":54,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032085566","type":"print"},{"value":"9783032085573","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T00:00:00Z","timestamp":1761696000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T00:00:00Z","timestamp":1761696000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-08557-3_5","type":"book-chapter","created":{"date-parts":[[2025,10,28]],"date-time":"2025-10-28T04:42:20Z","timestamp":1761626540000},"page":"50-65","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Semi-supervised Scene Text Detection based on Teacher-Student Scheme and Cascaded Hybrid Network"],"prefix":"10.1007","author":[{"given":"Fuchen","family":"Ma","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Songliang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinfu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yirui","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,29]]},"reference":[{"key":"5_CR1","doi-asserted-by":"crossref","unstructured":"Ando, A., Gidaris, S., Bursuc, A., Puy, G., Boulch, A., Marlet, R.: RangeViT: towards vision transformers for 3D semantic segmentation in autonomous driving. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 5240\u20135250 (2023)","DOI":"10.1109\/CVPR52729.2023.00507"},{"key":"5_CR2","doi-asserted-by":"crossref","unstructured":"Aralikatte, R., Abdou, M., Lent, H.C., Hershcovich, D., Sgaard, A.: Joint semantic analysis with document-level cross-task coherence rewards. In: Proceedings of AAAI Conference on Artificial Intelligence, pp. 12516\u201312525 (2021)","DOI":"10.1609\/aaai.v35i14.17484"},{"key":"5_CR3","doi-asserted-by":"crossref","unstructured":"Su, Y., et al.: LRANet: towards accurate and efficient scene text detection with low-rank approximation network, pp. 4979\u20134987. AAAI Press (2024)","DOI":"10.1609\/aaai.v38i5.28302"},{"key":"5_CR4","doi-asserted-by":"crossref","unstructured":"Wang, C.Y., Bochkovskiy, A., Liao, H.Y.M.: YOLOv7: trainable bag-of-freebies sets new state-of-the-art for real-time object detectors. In: Proceedings of The IEEE Conference on Computer Vision and Pattern Recognition, pp. 7464\u20137475 (2023)","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"5_CR5","unstructured":"Axelrod, B., Garg, S., Sharan, V., Valiant, G.: Sample Amplification: increasing dataset size even when learning is impossible. In: Proceedings of International Conference on Machine Learning, pp. 442\u2013451 (2020)"},{"key":"5_CR6","unstructured":"Hao, Y., Orlitsky, A.: Data Amplification: instance-optimal property estimation. In: Proceedings of International Conference on Machine Learning, pp. 4049\u20134059 (2020)"},{"issue":"9","key":"5_CR7","doi-asserted-by":"publisher","first-page":"6040","DOI":"10.1109\/TPAMI.2024.3379828","volume":"46","author":"W Yu","year":"2024","unstructured":"Yu, W., Liu, Y., Zhu, X., Cao, H., Sun, X., Bai, X.: Turning a CLIP model into a scene text spotter. IEEE Trans. Pattern Anal. Mach. Intell. 46(9), 6040\u20136054 (2024)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-015-0823-z","volume":"116","author":"M Jaderberg","year":"2016","unstructured":"Jaderberg, M., Simonyan, K., Vedaldi, A., Zisserman, A.: Reading text in the wild with convolutional neural networks. Int. J. Comput. Vision 116, 1\u201320 (2016)","journal-title":"Int. J. Comput. Vision"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Duan, C., Fu, P., Guo, S., Jiang, Q., Wei, X.: ODM: a text-image further alignment pre-training approach for scene text detection and spotting. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, pp. 15587\u201315597. IEEE (2024)","DOI":"10.1109\/CVPR52733.2024.01476"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Zheng, J., Fan, H., Zhang, L.: Kernel adaptive convolution for scene text detection via distance map prediction. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16-22, 2024, pp. 5957\u20135966. IEEE (2024)","DOI":"10.1109\/CVPR52733.2024.00569"},{"issue":"11","key":"5_CR11","first-page":"12908","volume":"45","author":"C Xue","year":"2023","unstructured":"Xue, C., Huang, J., Zhang, W., Lu, S., Wang, C., Bai, S.: Image-to-character-to-word transformers for accurate scene text recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(11), 12908\u201312921 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"Tian, Z., Huang, W., He, T., He, P., Qiao, Y.: Detecting text in natural image with connectionist text proposal network. In: Proceedings of European Conference on Computer Vision, pp. 56\u201372 (2016)","DOI":"10.1007\/978-3-319-46484-8_4"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Liao, M., Shi, B., Bai, X., Wang, X., Liu, W.: Textboxes: a fast text detector with a single deep neural network. In: Proceedings of AAAI Conference on Artificial Intelligence, pp. 4161\u20134167 (2017)","DOI":"10.1609\/aaai.v31i1.11196"},{"issue":"8","key":"5_CR14","doi-asserted-by":"publisher","first-page":"3676","DOI":"10.1109\/TIP.2018.2825107","volume":"27","author":"M Liao","year":"2018","unstructured":"Liao, M., Shi, B., Bai, X.: Textboxes++: a single-shot oriented scene text detector. IEEE Trans. Image Process. 27(8), 3676\u20133690 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Yang, F., Li, X., Cheng, H., Guo, Y., Chen, L., Li, J.: Multi-scale bidirectional FCN for object skeleton extraction. In: Proceedings of AAAI Conference on Artificial Intelligence, pp. 7461\u20137468 (2018)","DOI":"10.1609\/aaai.v32i1.12288"},{"key":"5_CR16","doi-asserted-by":"crossref","unstructured":"Zhang, D., Zhang, H., Li, H., Hu, X.: RR-FCN: rotational region-based fully convolutional networks for object detection. In: Proceedings of Engineering Applications of Neural Networks, pp. 58\u201370 (2018)","DOI":"10.1007\/978-3-319-98204-5_5"},{"key":"5_CR17","unstructured":"Lyu, P., Yang, Z., Leng, X., Wu, X., Li, R., Shen, X.: 2D attentional irregular scene text recognizer. CoRR abs\/1906.05708 (2019)"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Long, S., Ruan, J., Zhang, W., He, X., Wu, W., Yao, C.: TextSnake: a flexible representation for detecting text of arbitrary shapes. In: Proceedings of European Conference on Computer Vision, pp. 19\u201335 (2018)","DOI":"10.1007\/978-3-030-01216-8_2"},{"key":"5_CR19","doi-asserted-by":"crossref","unstructured":"Miyato, T., ichi Maeda, S., Koyama, M., Ishii, S.: Virtual adversarial training: a regularization method for supervised and semi-supervised learning. IEEE Trans. Pattern Anal. Mach. Intell. 41(8), 1979\u20131993 (2019)","DOI":"10.1109\/TPAMI.2018.2858821"},{"key":"5_CR20","doi-asserted-by":"crossref","unstructured":"Abuduweili, A., Li, X., Shi, H., Xu, C.Z., Dou, D.: Adaptive consistency regularization for semi-supervised transfer learning. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6923\u20136932 (2021)","DOI":"10.1109\/CVPR46437.2021.00685"},{"key":"5_CR21","unstructured":"Sohn, K., et al.: FixMatch: simplifying semi-supervised learning with consistency and confidence. In: Proceedings of Conference on Neural Information Processing Systems (2020)"},{"key":"5_CR22","doi-asserted-by":"crossref","unstructured":"Xie, Q., Luong, M.T., Hovy, E.H., Le, Q.V.: Self-training with noisy student improves imagenet classification. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2020)","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"5_CR23","unstructured":"Xie, Q., Dai, Z., Hovy, E.H., Luong, T., Le, Q.: Unsupervised data augmentation for consistency training. In: Proceedings of Conference on Neural Information Processing Systems (2020)"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Wang, Q., Li, W., Gool, L.V.: Semi-supervised learning by augmented distribution alignment. In: Proceedings of IEEE International Conference on Computer Vision, pp. 1466\u20131475 (2019)","DOI":"10.1109\/ICCV.2019.00155"},{"issue":"5","key":"5_CR25","doi-asserted-by":"publisher","first-page":"3306","DOI":"10.1109\/TII.2020.3036164","volume":"18","author":"X Zhang","year":"2022","unstructured":"Zhang, X., Wang, Z., Du, B.: Deep dynamic interest learning with session local and global consistency for click-through rate predictions. IEEE Trans. Ind. Info. 18(5), 3306\u20133315 (2022)","journal-title":"IEEE Trans. Ind. Info."},{"key":"5_CR26","unstructured":"Berthelot, D., Carlini, N., Goodfellow, I.J., Papernot, N., Oliver, A., Raffel, C.: MixMatch: a holistic approach to semi-supervised learning. In: Proceedings of Conference on Neural Information Processing Systems, pp. 5050\u20135060 (2019)"},{"key":"5_CR27","doi-asserted-by":"crossref","unstructured":"Huo, C., Jin, D., Li, Y., He, D., Yang, Y.B., Wu, L.: T2-GNN: graph neural networks for graphs with incomplete features and structure via teacher-student distillation. In: Proceedings of AAAI Conference on Artificial Intelligence, pp. 4339\u20134346 (2023)","DOI":"10.1609\/aaai.v37i4.25553"},{"key":"5_CR28","doi-asserted-by":"crossref","unstructured":"Iscen, A., Tolias, G., Avrithis, Y., Chum, O.: Label propagation for deep semi-supervised learning. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 5070\u20135079 (2019)","DOI":"10.1109\/CVPR.2019.00521"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Kim, J., et al.: Conmatch: Semi-supervised learning with confidence-guided consistency regularization. In: Proceedings of European Conference on Computer Vision, pp. 674\u2013690 (2022)","DOI":"10.1007\/978-3-031-20056-4_39"},{"key":"5_CR30","unstructured":"Chen, K., et al.: Hybrid task cascade for instance segmentation. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 4974\u20134983 (2019)"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Lin, T., Doll\u00e1r, P., Girshick, R.B., He, K., Hariharan, B., Belongie, S.J.: Feature pyramid networks for object detection. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017, Honolulu, HI, USA, July 21-26, 2017, pp. 936\u2013944. IEEE Computer Society (2017)","DOI":"10.1109\/CVPR.2017.106"},{"issue":"2","key":"5_CR32","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1016\/0167-8655(82)90019-8","volume":"1","author":"L Kitchen","year":"1982","unstructured":"Kitchen, L., Rosenfeld, A.: Non-maximum suppression of gradient magnitudes makes them easier to threshold. Pattern Recognit. Lett. 1(2), 93\u201394 (1982)","journal-title":"Pattern Recognit. Lett."},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Karatzas, D., et al.: ICDAR 2013 robust reading competition. In: Proceedings of International Conference on Document Analysis and Recognition, pp. 1484\u20131493 (2013)","DOI":"10.1109\/ICDAR.2013.221"},{"key":"5_CR34","doi-asserted-by":"crossref","unstructured":"Karatzas, D., et al.: ICDAR 2015 competition on robust reading. In: Proceedings of International Conference on Document Analysis and Recognition, pp. 1156\u20131160 (2015)","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"5_CR35","doi-asserted-by":"crossref","unstructured":"Nayef, N., et al.: ICDAR2017 robust reading challenge on multi-lingual scene text detection and script identification - RRC-MLT. In: Proceedings of International Conference on Document Analysis and Recognition, pp. 1454\u20131459 (2017)","DOI":"10.1109\/ICDAR.2017.237"},{"key":"5_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, C., Liang, B., Huang, Z., En, M., Han, J., Ding, E., Ding, X.: Look more than once: an accurate detector for text of arbitrary shapes. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 10552\u201310561 (2019)","DOI":"10.1109\/CVPR.2019.01080"},{"key":"5_CR37","doi-asserted-by":"crossref","unstructured":"von Braun, M.S., Frenzel, P., ding, C.K., Fuchs, M.: Utilizing mask R-CNN for waterline detection in canoe sprint video analysis. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 3826\u20133835 (2020)","DOI":"10.1109\/CVPRW50498.2020.00446"},{"key":"5_CR38","unstructured":"Zhang, B., Dong, J., Zhao, Z., Meng, Z., Su, F.: MT2: multi-task mean teacher for semi-supervised cell segmentation. In: Proceedings of Neural Information Processing Systems, pp. 1\u201313 (2022)"},{"key":"5_CR39","doi-asserted-by":"crossref","unstructured":"Shi, B., Bai, X., Belongie, S.J.: Detecting oriented text in natural images by linking segments. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 3482\u20133490 (2017)","DOI":"10.1109\/CVPR.2017.371"},{"key":"5_CR40","doi-asserted-by":"crossref","unstructured":"He, W., Zhang, X.Y., Yin, F., Liu, C.L.: Deep direct regression for multi-oriented scene text detection. In: Proceedings of International Conference on Computer Vision, pp. 745\u2013753 (2017)","DOI":"10.1109\/ICCV.2017.87"},{"key":"5_CR41","doi-asserted-by":"crossref","unstructured":"Deng, D., Liu, H., Li, X., Cai, D.: PixelLink: detecting scene text via instance segmentation. In: Proceedings of AAAI Conference on Artificial Intelligence, pp. 6773\u20136780 (2018)","DOI":"10.1609\/aaai.v32i1.12269"},{"key":"5_CR42","doi-asserted-by":"crossref","unstructured":"He, P., Huang, W., He, T., Zhu, Q., Qiao, Y., Li, X.: Single shot text detector with regional attention. In: Proceedings of International Conference on Computer Vision, pp. 3066\u20133074 (2017)","DOI":"10.1109\/ICCV.2017.331"},{"key":"5_CR43","doi-asserted-by":"crossref","unstructured":"Hu, H., Zhang, C., Luo, Y., Wang, Y., Han, J., Ding, E.: WordSup: exploiting word annotations for character based text detection. In: Proceedings of International Conference on Computer Vision, pp. 4950\u20134959 (2017)","DOI":"10.1109\/ICCV.2017.529"},{"key":"5_CR44","doi-asserted-by":"crossref","unstructured":"Liao, M., Zhu, Z., Shi, B., Xia, G.S., Bai, X.: Rotation-sensitive regression for oriented scene text detection. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 5909\u20135918 (2018)","DOI":"10.1109\/CVPR.2018.00619"},{"key":"5_CR45","doi-asserted-by":"crossref","unstructured":"Zhou, X., et al.: East: An efficient and accurate scene text detector. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 2642\u20132651 (2017)","DOI":"10.1109\/CVPR.2017.283"},{"key":"5_CR46","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Chen, J., Liang, L., Kuang, Z., Jin, L., Zhang, W.: Fourier contour embedding for arbitrary-shaped text detection. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 3123\u20133131 (2021)","DOI":"10.1109\/CVPR46437.2021.00314"},{"issue":"11","key":"5_CR47","doi-asserted-by":"publisher","first-page":"5406","DOI":"10.1109\/TIP.2018.2855399","volume":"27","author":"W He","year":"2018","unstructured":"He, W., Zhang, X.Y., Yin, F., Liu, C.L.: Multi-oriented and multi-lingual scene text detection with direct regression. IEEE Trans. Image Process. 27(11), 5406\u20135419 (2018)","journal-title":"IEEE Trans. Image Process."},{"issue":"2","key":"5_CR48","doi-asserted-by":"publisher","first-page":"532","DOI":"10.1109\/TPAMI.2019.2937086","volume":"43","author":"M Liao","year":"2021","unstructured":"Liao, M., Lyu, P., He, M., Yao, C., Wu, W., Bai, X.: Mask TextSpotter: an end-to-end trainable neural network for spotting text with arbitrary shapes. IEEE Trans. Pattern Anal. Mach. Intell. 43(2), 532\u2013548 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5_CR49","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Hou, W., Lu, T., Yu, G., Shao, S.: Shape robust text detection with progressive scale expansion network. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition. pp. 9336\u20139345 (2019)","DOI":"10.1109\/CVPR.2019.00956"},{"key":"5_CR50","doi-asserted-by":"crossref","unstructured":"Baek, Y., Lee, B., Han, D., Yun, S., Lee, H.: Character region awareness for text detection. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 9365\u20139374 (2019)","DOI":"10.1109\/CVPR.2019.00959"},{"key":"5_CR51","unstructured":"Liu, Y., Jin, L., Zhang, S., Zhang, S.: Detecting curve text in the wild: new dataset and new solution. CoRR abs\/1712.02170 (2017)"},{"key":"5_CR52","doi-asserted-by":"crossref","unstructured":"Wang, F., Chen, Y., Wu, F., Li, X.: TextRay: contour-based geometric modeling for arbitrary-shaped scene text detection. In: Proceedings of ACM Conference on Multimedia, pp. 111\u2013119 (2020)","DOI":"10.1145\/3394171.3413819"},{"key":"5_CR53","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, G., Yang, S., Liu, F., Lin, W., Goh, W.L.: Towards robust curve text detection with conditional spatial expansion. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, pp. 7269\u20137278 (2019)","DOI":"10.1109\/CVPR.2019.00744"},{"key":"5_CR54","doi-asserted-by":"crossref","unstructured":"Feng, W., He, W., Yin, F., Zhang, X.Y., Liu, C.L.: TextDragon: an end-to-end framework for arbitrary shaped text spotting. In: Proceedings of International Conference on Computer Vision, pp. 9075\u20139084 (2019)","DOI":"10.1109\/ICCV.2019.00917"}],"container-title":["Lecture Notes in Computer Science","AI and Multimodal Services \u2013 AIMS 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-08557-3_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T02:28:29Z","timestamp":1781490509000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-08557-3_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,29]]},"ISBN":["9783032085566","9783032085573"],"references-count":54,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-08557-3_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,29]]},"assertion":[{"value":"29 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AIMS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on AI and Multimodal Services","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aimse2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.servicessociety.org\/aims","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}