{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T00:05:55Z","timestamp":1780617955634,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":70,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["32341012, 62172103"],"award-info":[{"award-number":["32341012, 62172103"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681390","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"5191-5200","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Decoder Pre-Training with only Text for Scene Text Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3387-2293","authenticated-orcid":false,"given":"Shuai","family":"Zhao","sequence":"first","affiliation":[{"name":"School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3114-2188","authenticated-orcid":false,"given":"Yongkun","family":"Du","sequence":"additional","affiliation":[{"name":"School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1543-6889","authenticated-orcid":false,"given":"Zhineng","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1907-8567","authenticated-orcid":false,"given":"Yu-Gang","family":"Jiang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"A. Aberdam R. Litman S. Tsiper O. Anschel R. Slossberg S. Mazor R. Manmatha and P. Perona. 2021. Sequence-to-Sequence Contrastive Learning for Text Recognition. In CVPR. 15302--15312.","DOI":"10.1109\/CVPR46437.2021.01505"},{"key":"e_1_3_2_1_2_1","first-page":"18","article-title":"2014. A robust arbitrary text detection system for natural scene images","volume":"41","author":"Anhar R.","year":"2014","unstructured":"R. Anhar, S. Palaiahnakote, C. S. Chan, and C. L. Tan. 2014. A robust arbitrary text detection system for natural scene images. Expert Systems with Applications, Vol. 41, 18 (2014), 8027--8048.","journal-title":"Expert Systems with Applications"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"R. Atienza. 2021. Vision Transformer for Fast and Efficient Scene Text Recognition. In ICDAR. 319--334.","DOI":"10.1007\/978-3-030-86549-8_21"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"J. Baek Y. Matsui and K. Aizawa. 2021. What if we only use real datasets for scene text recognition? toward scene text recognition with fewer labels. In CVPR. 3113--3122.","DOI":"10.1109\/CVPR46437.2021.00313"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"D. Bautista and R.l Atienza. 2022. Scene Text Recognition with Permuted Autoregressive Sequence Models. In ECCV. 178--196.","DOI":"10.1007\/978-3-031-19815-1_11"},{"key":"e_1_3_2_1_6_1","volume-title":"Rosetta: Large Scale System for Text Detection and Recognition in Images. In ACM SIGKDD. 71--79.","author":"Borisyuk F.","year":"2018","unstructured":"F. Borisyuk, A. Gordo, and V. Sivakumar. 2018. Rosetta: Large Scale System for Text Detection and Recognition in Images. In ACM SIGKDD. 71--79."},{"key":"e_1_3_2_1_7_1","volume-title":"ACCV Workshops. 127--143","author":"Buvsta M.","unstructured":"M. Buvsta, Y. Patel, and J. Matas. 2019. E2e-mlt-an unconstrained end-to-end method for multi-language scene text. In ACCV Workshops. 127--143."},{"key":"e_1_3_2_1_8_1","unstructured":"H. Cai J. Sun and Y. Xiong. 2021. Revisiting Classification Perspective on Scene Text Recognition. arXiv:2102.10884 (2021)."},{"key":"e_1_3_2_1_9_1","unstructured":"F. Carlsson P. Eisen F. Rekathati and M. Sahlgren. 2022. Cross-lingual and multilingual clip. In LREC. 6848--6854."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"J. Chen B. Li and X. Xue. 2021. Scene Text Telescope: Text-Focused Scene Image Super-Resolution. In CVPR. 12021--12030.","DOI":"10.1109\/CVPR46437.2021.01185"},{"key":"e_1_3_2_1_11_1","unstructured":"J. Chen H. Yu J. Ma M. Guan X. Xu X. Wang S. Qu B. Li and X. Xue. 2021. Benchmarking Chinese Text Recognition: Datasets Baselines and an Empirical Study. arXiv:2112.15093 (2021)."},{"key":"e_1_3_2_1_12_1","unstructured":"T. Chen S. Kornblith M. Norouzi and G. Hinton. 2020. A simple framework for contrastive learning of visual representations. In ICML. 1597--1607."},{"key":"e_1_3_2_1_13_1","first-page":"1","article-title":"2023. LISTER: Neighbor decoding for length-insensitive scene text recognition","author":"Cheng C.","year":"1954","unstructured":"C. Cheng, P. Wang, C. Da, Q. Zheng, and C. Yao. 2023. LISTER: Neighbor decoding for length-insensitive scene text recognition. In ICCV. 19541--19551.","journal-title":"ICCV."},{"key":"e_1_3_2_1_14_1","volume-title":"ICDAR2019 Robust Reading Challenge on Arbitrary-Shaped Text - RRC-ArT. In ICDAR. 1571--1576","author":"Chng C.","unstructured":"C. Chng, E. Ding, J. Liu, D. Karatzas, C. Chan, L. Jin, Y. Liu, Y. Sun, C. Ng, C. Luo, Z. Ni, C. Fang, S. Zhang, and J. Han. 2019. ICDAR2019 Robust Reading Challenge on Arbitrary-Shaped Text - RRC-ArT. In ICDAR. 1571--1576."},{"key":"e_1_3_2_1_15_1","unstructured":"C. Da P. Wang and C. Yao. 2023. Multi-Granularity Prediction with Learnable Fusion for Scene Text Recognition. arXiv:2307.13244 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Words: Transformers for Image Recognition at Scale. In ICLR. 1--21.","author":"Dosovitskiy A.","year":"2021","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, and N. Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In ICLR. 1--21."},{"key":"e_1_3_2_1_17_1","unstructured":"Y. Du Z. Chen C. Jia X. Yin C. Li Y. Du and Y. Jiang. 2023. Context Perception Parallel Decoder for Scene Text Recognition. arXiv:2307.12270 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"SVTR: Scene Text Recognition with a Single Visual Model. In IJCAI. 884--890.","author":"Du Y.","year":"2022","unstructured":"Y. Du, Z. Chen, C. Jia, X. Yin, T. Zheng, C. Li, Y. Du, and Y. Jiang. 2022. SVTR: Scene Text Recognition with a Single Visual Model. In IJCAI. 884--890."},{"key":"e_1_3_2_1_19_1","unstructured":"Y. Du Z. Chen Y. Su C. Jia and Y. Jiang. 2024. Instruction-Guided Scene Text Recognition. arXiv:2401.17851 (2024)."},{"key":"e_1_3_2_1_20_1","first-page":"6","article-title":"2023. ABINet: Autonomous","volume":"45","author":"Fang S.","year":"2023","unstructured":"S. Fang, Z. Mao, H. Xie, Y. Wang, C. Yan, and Y. Zhang. 2023. ABINet: Autonomous, Bidirectional and Iterative Language Modeling for Scene Text Spotting. IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 45, 6 (2023), 7123--7141.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"S. Fang H. Xie Y. Wang Z. Mao and Y. Zhang. 2021. Read Like Humans: Autonomous Bidirectional and Iterative Language Modeling for Scene Text Recognition. In CVPR. 7098--7107.","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"A. Graves S. Fern\u00e1ndez F. Gomez and J. Schmidhuber. 2006. Connectionist Temporal Classification: Labelling Unsegmented Sequence Data with Recurrent Neural Networks. In ICML. 369--376.","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"T. Guan C. Gu J. Tu X. Yang Q. Feng Y. Zhao and W. Shen. 2023. Self-supervised implicit glyph attention for text recognition. In CVPR. 15285--15294.","DOI":"10.1109\/CVPR52729.2023.01467"},{"key":"e_1_3_2_1_24_1","first-page":"3","article-title":"2023. Self-supervised character-to-character distillation for text recognition","author":"Guan T.","year":"1947","unstructured":"T. Guan, W. Shen, X. Yang, Q. Feng, Z. Jiang, and X. Yang. 2023. Self-supervised character-to-character distillation for text recognition. In ICCV. 19473--19484.","journal-title":"ICCV."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"A. Gupta A. Vedaldi and A. Zisserman. 2016. Synthetic Data for Text Localisation in Natural Images. In CVPR. 2315--2324.","DOI":"10.1109\/CVPR.2016.254"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"K. He X. Chen S. Xie Y. Li P. Doll\u00e1r and R. Girshick. 2022. Masked autoencoders are scalable vision learners. In CVPR. 16000--16009.","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"K. He X. Zhang S. Ren and J. Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Y. He C. Chen J. Zhang J. Liu F. He C. Wang and B. Du. 2022. Visual Semantics Allow for Textual Reasoning Better in Scene Text Recognition. In AAAI. 888--896.","DOI":"10.1609\/aaai.v36i1.19971"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0823-z"},{"key":"e_1_3_2_1_30_1","first-page":"3","article-title":"2023. Revisiting scene text recognition: A data perspective","author":"Jiang Q.","year":"2054","unstructured":"Q. Jiang, J. Wang, D. Peng, C. Liu, and L. Jin. 2023. Revisiting scene text recognition: A data perspective. In ICCV. 20543--20554.","journal-title":"ICCV."},{"key":"e_1_3_2_1_31_1","volume-title":"ICDAR 2015 competition on Robust Reading. In ICDAR. 1156--1160","author":"Karatzas D.","unstructured":"D. Karatzas, L. Gomez-Bigorda, A. Nicolaou, S. Ghosh, A. Bagdanov, M. Iwamura, J. Matas, L. Neumann, V. R. Chandrasekhar, S. Lu, F. Shafait, S. Uchida, and E. Valveny. 2015. ICDAR 2015 competition on Robust Reading. In ICDAR. 1156--1160."},{"key":"e_1_3_2_1_32_1","volume-title":"ICDAR 2013 Robust Reading Competition. In ICDAR. 1484--1493","author":"Karatzas D.","unstructured":"D. Karatzas, F. Shafait, S. Uchida, M. Iwamura, L. G. i. Bigorda, S. R. Mestre, J. Mas, D. F. Mota, J. A. Almaz\u00e0n, and L. P. de las Heras. 2013. ICDAR 2013 Robust Reading Competition. In ICDAR. 1484--1493."},{"key":"e_1_3_2_1_33_1","volume-title":"Openimages: A public dataset for large-scale multi-label and multi-class image classification. Dataset available from https:\/\/github. com\/openimages","author":"Krasin I.","year":"2017","unstructured":"I. Krasin, T. Duerig, N. Alldrin, V. Ferrari, S. Abu-El-Haija, A. Kuznetsova, H. Rom, J. Uijlings, S. Popov, A. Veit, et al. 2017. Openimages: A public dataset for large-scale multi-label and multi-class image classification. Dataset available from https:\/\/github. com\/openimages, Vol. 2, 3 (2017), 18."},{"key":"e_1_3_2_1_34_1","unstructured":"I. Krylov S.K. Nosov and V. Sovrasov. 2021. Open images v5 text annotation and yet another mask text spotter. In ACML. PMLR 379--389."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"C. Lee and S. Osindero. 2016. Recursive recurrent nets with attention modeling for ocr in the wild. In CVPR. 2231--2239.","DOI":"10.1109\/CVPR.2016.245"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"H. Li P. Wang C. Shen and G. Zhang. 2019. Show attend and read: A simple and strong baseline for irregular text recognition. In AAAI. 8610--8617.","DOI":"10.1609\/aaai.v33i01.33018610"},{"key":"e_1_3_2_1_37_1","volume-title":"Trocr: Transformer-based optical character recognition with pre-trained models. In AAAI. 13094--13102.","author":"Li M.","year":"2023","unstructured":"M. Li, T. Lv, J. Chen, L. Cui, Y. Lu, D. Florencio, C. Zhang, Z. Li, and F. Wei. 2023. Trocr: Transformer-based optical character recognition with pre-trained models. In AAAI. 13094--13102."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"T. Lin M. Maire S. Belongie J. Hays P. Perona D. Ramanan P. Doll\u00e1r and C. Zitnick. 2014. Microsoft coco: Common objects in context. In ECCV. 740--755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107980"},{"key":"e_1_3_2_1_40_1","volume-title":"MORAN: A Multi-Object Rectified Attention Network for Scene Text Recognition. Pattern Recognition","author":"Luo C.","year":"2019","unstructured":"C. Luo, L. Jin, and Z. Sun. 2019. MORAN: A Multi-Object Rectified Attention Network for Scene Text Recognition. Pattern Recognition (2019), 109--118."},{"key":"e_1_3_2_1_41_1","unstructured":"P. Lyu C. Zhang S. Liu M. Qiao Y. Xu L. Wu K. Yao J. Han E. Ding and J. Wang. 2022. Maskocr: text recognition with masked encoder-decoder pretraining. arXiv:2206.00311 (2022)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"A. Mishra A. Karteek and C. V. Jawahar. 2012. Scene Text Recognition using Higher Order Language Priors. In BMVC. 1--11.","DOI":"10.5244\/C.26.127"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00254"},{"key":"e_1_3_2_1_44_1","volume-title":"ICDAR2017 robust reading challenge on multi-lingual scene text detection and script identification-rrc-mlt. In ICDAR. 1454--1459","author":"Nayef N.","year":"2017","unstructured":"N. Nayef, F. Yin, I. Bizid, H. Choi, Y. Feng, D. Karatzas, Z. Luo, U. Pal, C. Rigaud, J. Chazalon, et al. 2017. ICDAR2017 robust reading challenge on multi-lingual scene text detection and script identification-rrc-mlt. In ICDAR. 1454--1459."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"T. Q. Phan P. Shivakumara S. Tian and C. L. Tan. 2013. Recognizing Text with Perspective Distortion in Natural Scenes. In CVPR. 569--576.","DOI":"10.1109\/ICCV.2013.76"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Z. Qiao Y. Zhou J. Wei W. Wang Y. Zhang N. Jiang H. Wang and W. Wang. 2021. Pimnet: a parallel iterative and mimicking network for scene text recognition. In ACM MM. 2046--2055.","DOI":"10.1145\/3474085.3475238"},{"key":"e_1_3_2_1_47_1","volume-title":"SEED: Semantics Enhanced Encoder-Decoder Framework for Scene Text Recognition. In CVPR. 13525--13534.","author":"Qiao Z.","year":"2020","unstructured":"Z. Qiao, Y. Zhou, D. Yang, Y. Zhou, and W. Wang. 2020. SEED: Semantics Enhanced Encoder-Decoder Framework for Scene Text Recognition. In CVPR. 13525--13534."},{"key":"e_1_3_2_1_48_1","unstructured":"A. Radford J. Kim C. Hallacy A. Ramesh G. Goh S. Agarwal G. Sastry A. Askell P. Mishkin J. Clark G. Krueger and I. Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In ICML. PMLR 8748--8763."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"M. Rang Z. Bi C. Liu Y. Wang and K. Han. 2024. An Empirical Study of Scaling Law for Scene Text Recognition. In CVPR. 15619--15629.","DOI":"10.1109\/CVPR52733.2024.01479"},{"key":"e_1_3_2_1_50_1","volume-title":"NRTR: A No-Recurrence Sequence-to-Sequence Model for Scene Text Recognition. In ICDAR. 781--786.","author":"Sheng F.","year":"2019","unstructured":"F. Sheng, Z. Chen, and B. Xu. 2019. NRTR: A No-Recurrence Sequence-to-Sequence Model for Scene Text Recognition. In ICDAR. 781--786."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2646371"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2848939"},{"key":"e_1_3_2_1_53_1","volume-title":"ICDAR2017 Competition on Reading Chinese Text in the Wild (RCTW-17)","author":"Shi B.","unstructured":"B. Shi, C. Yao, M. Liao, M. Yang, P. Xu, L. Cui, S. Belongie, S. Lu, and X. Bai. 2017. ICDAR2017 Competition on Reading Chinese Text in the Wild (RCTW-17). In ICDAR. 1429--1434."},{"key":"e_1_3_2_1_54_1","volume-title":"Textocr: Towards large-scale end-to-end reasoning for arbitrary-shaped scene text. In CVPR. 8802--8812.","author":"Singh A.","year":"2021","unstructured":"A. Singh, G. Pang, M. Toh, J. Huang, W.h Galuba, and T. Hassner. 2021. Textocr: Towards large-scale end-to-end reasoning for arbitrary-shaped scene text. In CVPR. 8802--8812."},{"key":"e_1_3_2_1_55_1","volume-title":"ICDAR 2019 Competition on Large-Scale Street View Text with Partial Labeling - RRC-LSVT. In ICDAR. 1557--1562","author":"Sun Y.","unstructured":"Y. Sun, D. Karatzas, C. Chan, L. Jin, Z. Ni, C. Chng, Y. Liu, C. Luo, C. Ng, J. Han, E. Ding, and J. Liu. 2019. ICDAR 2019 Competition on Large-Scale Street View Text with Partial Labeling - RRC-LSVT. In ICDAR. 1557--1562."},{"key":"e_1_3_2_1_56_1","unstructured":"A. Veit T. Matera L.s Neumann J. Matas and S. Belongie. 2016. COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images. arXiv (2016)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"crossref","unstructured":"K. Wang B. Babenko and S. Belongie. 2011. End-to-end scene text recognition. In ICCV. 1457--1464.","DOI":"10.1109\/ICCV.2011.6126402"},{"key":"e_1_3_2_1_58_1","volume-title":"One: A New Scene Text Recognizer With Visual Language Modeling Network. In ICCV. 14194--14203.","author":"Wang Y.","year":"2021","unstructured":"Y. Wang, H. Xie, S. Fang, J. Wang, S. Zhu, and Y. Zhang. 2021. From Two to One: A New Scene Text Recognizer With Visual Language Modeling Network. In ICCV. 14194--14203."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3197981"},{"key":"e_1_3_2_1_60_1","volume-title":"OTE: Exploring Accurate Scene Text Recognition Using One Token. In CVPR. 28327--28336.","author":"Xu J.","year":"2024","unstructured":"J. Xu, Y. Wang, H. Xie, and Y. Zhang. 2024. OTE: Exploring Accurate Scene Text Recognition Using One Token. In CVPR. 28327--28336."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"M. Yang M. Liao P. Lu J. Wang S. Zhu H. Luo Q. Tian and X. Bai. 2022. Reading and writing: Discriminative and generative modeling for self-supervised text recognition. In ACM MM. 4214--4223.","DOI":"10.1145\/3503161.3547784"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"crossref","unstructured":"D. Yu X. Li C. Zhang T. Liu J. Han J. Liu and E. Ding. 2020. Towards Accurate Scene Text Recognition With Semantic Reasoning Networks. In CVPR. 12110--12119.","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"crossref","unstructured":"H. Yu X. Wang B. Li and X. Xue. 2023. Chinese Text Recognition with A Pre-Trained CLIP-Like Model Through Image-IDS Aligning. In ICCV. 11943--11952.","DOI":"10.1109\/ICCV51070.2023.01097"},{"key":"e_1_3_2_1_64_1","volume-title":"Linguistic More: Taking a Further Step toward Efficient and Accurate Scene Text Recognition. In IJCAI. 1704--1712.","author":"Zhang B.","year":"2023","unstructured":"B. Zhang, H. Xie, Y. Wang, J. Xu, and Y. Zhang. 2023. Linguistic More: Taking a Further Step toward Efficient and Accurate Scene Text Recognition. In IJCAI. 1704--1712."},{"key":"e_1_3_2_1_65_1","volume-title":"ICDAR 2019 Robust Reading Challenge on Reading Chinese Text on Signboard. In ICDAR. 1577--1581","author":"Zhang R.","unstructured":"R. Zhang, M. Yang, B. Xiang, B. Shi, K. Dimosthenis, S. Lu, C. Jawahar, Zhou Y., Q. Jiang, S. Qi, N. Li, Z. Kai, L. Wang, D. Wang, and M. Liao. 2019. ICDAR 2019 Robust Reading Challenge on Reading Chinese Text on Signboard. In ICDAR. 1577--1581."},{"key":"e_1_3_2_1_66_1","volume-title":"SUNw: Scene Understanding Workshop-CVPR. 5.","author":"Zhang Y.","unstructured":"Y. Zhang, L. Gueguen, I. Zharkov, P. Zhang, K. Seifert, and B. Kadlec. 2017. Uber-text: A large-scale dataset for optical character recognition from street-level imagery. In SUNw: Scene Understanding Workshop-CVPR. 5."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"crossref","unstructured":"Z. Zhao J. Tang C. Lin B. Wu C. Huang H. Liu X. Tan Z. Zhang and Y. Xie. 2024. Multi-modal In-Context Learning Makes an Ego-evolving Scene Text Recognizer. In CVPR. 15567--15576.","DOI":"10.1109\/CVPR52733.2024.01474"},{"key":"e_1_3_2_1_68_1","volume-title":"TPS: Attention-Enhanced Thin-Plate Spline for Scene Text Recognition. In IJCAI. 1777--1785.","author":"Zheng T.","year":"2023","unstructured":"T. Zheng, Z. Chen, J. Bai, H. Xie, and Y. Jiang. 2023. TPS: Attention-Enhanced Thin-Plate Spline for Scene Text Recognition. In IJCAI. 1777--1785."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01880-0"},{"key":"e_1_3_2_1_70_1","volume-title":"MRN: Multiplexed routing network for incremental multilingual text recognition. In ICCV. 18644--18653.","author":"Zheng T.","year":"2023","unstructured":"T. Zheng, Z. Chen, B. Huang, W. Zhang, and Y. Jiang. 2023. MRN: Multiplexed routing network for incremental multilingual text recognition. In ICCV. 18644--18653."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681390","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681390","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:44Z","timestamp":1750295864000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681390"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":70,"alternative-id":["10.1145\/3664647.3681390","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681390","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}