{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:48:39Z","timestamp":1777657719986,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T00:00:00Z","timestamp":1729468800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,21]]},"DOI":"10.1145\/3627673.3679618","type":"proceedings-article","created":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T19:34:11Z","timestamp":1729452851000},"page":"365-373","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["SVIPTR: Fast and Efficient Scene Text Recognition with Vision Permutable Extractor"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1130-8302","authenticated-orcid":false,"given":"Xianfu","family":"Cheng","sequence":"first","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8929-0834","authenticated-orcid":false,"given":"Weixiao","family":"Zhou","sequence":"additional","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2003-9217","authenticated-orcid":false,"given":"Xiang","family":"Li","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1983-012X","authenticated-orcid":false,"given":"Jian","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3310-079X","authenticated-orcid":false,"given":"Hang","family":"Zhang","sequence":"additional","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8694-8530","authenticated-orcid":false,"given":"Tao","family":"Sun","sequence":"additional","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4419-4551","authenticated-orcid":false,"given":"Wei","family":"Zhang","sequence":"additional","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3448-1746","authenticated-orcid":false,"given":"Yuying","family":"Mai","sequence":"additional","affiliation":[{"name":"Beijing Jiaotong University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2488-2787","authenticated-orcid":false,"given":"Tongliang","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing Information Science and Technology University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9314-3753","authenticated-orcid":false,"given":"Xiaoming","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen Intelligent Strong Technology Co., Ltd., Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9603-9713","authenticated-orcid":false,"given":"Zhoujun","family":"Li","sequence":"additional","affiliation":[{"name":"CCSE, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,21]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Rowel Atienza. 2021. Vision transformer for fast and efficient scene text recognition. In ICDAR.","DOI":"10.1007\/978-3-030-86549-8_21"},{"key":"e_1_3_2_1_2_1","volume-title":"Rosetta: Large scale system for text detection and recognition in images. In KDD.","author":"Borisyuk Fedor","year":"2018","unstructured":"Fedor Borisyuk, Albert Gordo, and Viswanath Sivakumar. 2018. Rosetta: Large scale system for text detection and recognition in images. In KDD."},{"key":"e_1_3_2_1_3_1","volume-title":"LISTER: Neighbor decoding for length-insensitive scene text recognition. In ICCV.","author":"Cheng Changxu","year":"2023","unstructured":"Changxu Cheng, Peng Wang, Cheng Da, Qi Zheng, and Cong Yao. 2023. LISTER: Neighbor decoding for length-insensitive scene text recognition. In ICCV."},{"key":"e_1_3_2_1_4_1","first-page":"10694","article-title":"An improved stochastic gradient descent algorithm based on R\u00e9nyi differential privacy","volume":"37","author":"Cheng XianFu","year":"2022","unstructured":"XianFu Cheng, YanQing Yao, Liying Zhang, Ao Liu, and Zhoujun Li. 2022. An improved stochastic gradient descent algorithm based on R\u00e9nyi differential privacy. IJIS, Vol. 37, 12 (2022), 10694--10714.","journal-title":"IJIS"},{"key":"e_1_3_2_1_5_1","volume-title":"Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882","author":"Chu Xiangxiang","year":"2021","unstructured":"Xiangxiang Chu, Zhi Tian, Bo Zhang, Xinlong Wang, and Chunhua Shen. 2021. Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882 (2021)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Xiaoyi Dong Jianmin Bao Dongdong Chen Weiming Zhang Nenghai Yu Lu Yuan Dong Chen and Baining Guo. 2022. Cswin transformer: A general vision transformer backbone with cross-shaped windows. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"e_1_3_2_1_7_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_8_1","volume-title":"Svtr: Scene text recognition with a single visual model. arXiv preprint arXiv:2205.00159","author":"Du Yongkun","year":"2022","unstructured":"Yongkun Du, Zhineng Chen, Caiyan Jia, Xiaoting Yin, Tianlun Zheng, Chenxia Li, Yuning Du, and Yu-Gang Jiang. 2022. Svtr: Scene text recognition with a single visual model. arXiv preprint arXiv:2205.00159 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Rmt: Retentive networks meet vision transformers. In CVPR.","author":"Fan Qihang","year":"2024","unstructured":"Qihang Fan, Huaibo Huang, Mingrui Chen, Hongmin Liu, and Ran He. 2024. Rmt: Retentive networks meet vision transformers. In CVPR."},{"key":"e_1_3_2_1_10_1","unstructured":"Shancheng Fang Hongtao Xie Yuxin Wang Zhendong Mao and Yongdong Zhang. 2021. Read like humans: Autonomous bidirectional and iterative language modeling for scene text recognition. In CVPR."},{"key":"e_1_3_2_1_11_1","volume-title":"Dtrocr: Decoder-only transformer for optical character recognition. In WACV.","author":"Fujitake Masato","year":"2024","unstructured":"Masato Fujitake. 2024. Dtrocr: Decoder-only transformer for optical character recognition. In WACV."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Raul Gomez Baoguang Shi Lluis Gomez Lukas Numann Andreas Veit Jiri Matas Serge Belongie and Dimosthenis Karatzas. 2017. Icdar2017 robust reading challenge on coco-text. In ICDAR.","DOI":"10.1109\/ICDAR.2017.234"},{"key":"e_1_3_2_1_13_1","volume-title":"Lvp-m3: language-aware visual prompt for multilingual multimodal machine translation. arXiv preprint arXiv:2210.15461","author":"Guo Hongcheng","year":"2022","unstructured":"Hongcheng Guo, Jiaheng Liu, Haoyang Huang, Jian Yang, Zhoujun Li, Dongdong Zhang, Zheng Cui, and Furu Wei. 2022. Lvp-m3: language-aware visual prompt for multilingual multimodal machine translation. arXiv preprint arXiv:2210.15461 (2022)."},{"key":"e_1_3_2_1_14_1","volume-title":"M2C: towards automatic multimodal manga complement. arXiv preprint arXiv:2310.17130","author":"Guo Hongcheng","year":"2023","unstructured":"Hongcheng Guo, Boyang Wang, Jiaqi Bai, Jiaheng Liu, Jian Yang, and Zhoujun Li. 2023. M2C: towards automatic multimodal manga complement. arXiv preprint arXiv:2310.17130 (2023)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Ankush Gupta Andrea Vedaldi and Andrew Zisserman. 2016. Synthetic data for text localisation in natural images. In CVPR.","DOI":"10.1109\/CVPR.2016.254"},{"key":"e_1_3_2_1_16_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR."},{"key":"e_1_3_2_1_17_1","volume-title":"Gtc: Guided training of ctc towards efficient and accurate scene text recognition. In AAAI.","author":"Hu Wenyang","year":"2020","unstructured":"Wenyang Hu, Xiaocong Cai, Jun Hou, Shuai Yi, and Zhiping Lin. 2020. Gtc: Guided training of ctc towards efficient and accurate scene text recognition. In AAAI."},{"key":"e_1_3_2_1_18_1","volume-title":"Synthetic data and artificial neural networks for natural scene text recognition. arXiv preprint arXiv:1406.2227","author":"Jaderberg Max","year":"2014","unstructured":"Max Jaderberg, Karen Simonyan, Andrea Vedaldi, and Andrew Zisserman. 2014. Synthetic data and artificial neural networks for natural scene text recognition. arXiv preprint arXiv:1406.2227 (2014)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0823-z"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.221"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00281"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Hui Li Peng Wang Chunhua Shen and Guyu Zhang. 2019. Show attend and read: A simple and strong baseline for irregular text recognition. In AAAI.","DOI":"10.1609\/aaai.v33i01.33018610"},{"key":"e_1_3_2_1_24_1","first-page":"4094","article-title":"Dual relation network for scene text recognition","volume":"25","author":"Li Ming","year":"2022","unstructured":"Ming Li, Bin Fu, Han Chen, Junjun He, and Yu Qiao. 2022. Dual relation network for scene text recognition. IEEE TMM, Vol. 25 (2022), 4094--4107.","journal-title":"IEEE TMM"},{"key":"e_1_3_2_1_25_1","volume-title":"Trocr: Transformer-based optical character recognition with pre-trained models. In AAAI.","author":"Li Minghao","year":"2023","unstructured":"Minghao Li, Tengchao Lv, Jingye Chen, Lei Cui, Yijuan Lu, Dinei Florencio, Cha Zhang, Zhoujun Li, and Furu Wei. 2023. Trocr: Transformer-based optical character recognition with pre-trained models. In AAAI."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Ze Liu Yutong Lin Yue Cao Han Hu Yixuan Wei Zheng Zhang Stephen Lin and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_27_1","volume-title":"TransXNet: learning both global and local dynamics with a dual dynamic token mixer for visual recognition. arXiv preprint arXiv:2310.19380","author":"Lou Meng","year":"2023","unstructured":"Meng Lou, Hong-Yu Zhou, Sibei Yang, and Yizhou Yu. 2023. TransXNet: learning both global and local dynamics with a dual dynamic token mixer for visual recognition. arXiv preprint arXiv:2310.19380 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"Maskocr: Text recognition with masked encoder-decoder pretraining. arXiv preprint arXiv:2206.00311","author":"Lyu Pengyuan","year":"2022","unstructured":"Pengyuan Lyu, Chengquan Zhang, Shanshan Liu, Meina Qiao, Yangliu Xu, Liang Wu, Kun Yao, Junyu Han, Errui Ding, and Jingdong Wang. 2022. Maskocr: Text recognition with masked encoder-decoder pretraining. arXiv preprint arXiv:2206.00311 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Anand Mishra Karteek Alahari and CV Jawahar. 2012. Scene text recognition using higher order language priors. In BMVC.","DOI":"10.5244\/C.26.127"},{"key":"e_1_3_2_1_30_1","unstructured":"Byeonghu Na Yoonsik Kim and Sungrae Park. 2022. Multi-modal text recognition networks: Interactive enhancements between visual and semantic features. In ECCV."},{"key":"e_1_3_2_1_31_1","unstructured":"Trung Quy Phan Palaiahnakote Shivakumara Shangxuan Tian and Chew Lim Tan. 2013. Recognizing text with perspective distortion in natural scenes. In ICCV."},{"key":"e_1_3_2_1_32_1","volume-title":"NRTR: A no-recurrence sequence-to-sequence model for scene text recognition. In ICDAR.","author":"Sheng Fenfen","year":"2019","unstructured":"Fenfen Sheng, Zhineng Chen, and Bo Xu. 2019. NRTR: A no-recurrence sequence-to-sequence model for scene text recognition. In ICDAR."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2646371"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2848939"},{"key":"e_1_3_2_1_35_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_36_1","unstructured":"Tao Sun Dongsu Shen Saiqin Long Qingyong Deng and Shiguo Wang. 2022. Neural Distinguishers on TinyJAMBU-128 and GIFT-64. In ICONIP."},{"key":"e_1_3_2_1_37_1","volume-title":"Visual-semantic transformer for scene text recognition. arXiv preprint arXiv:2112.00948","author":"Tang Xin","year":"2021","unstructured":"Xin Tang, Yongquan Lai, Ying Liu, Yuanyuan Fu, and Rui Fang. 2021. Visual-semantic transformer for scene text recognition. arXiv preprint arXiv:2112.00948 (2021)."},{"key":"e_1_3_2_1_38_1","unstructured":"Hugo Touvron Matthieu Cord Matthijs Douze Francisco Massa Alexandre Sablayrolles and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In ICML."},{"key":"e_1_3_2_1_39_1","volume-title":"Attention is all you need. Advances in Neural Information Processing Systems","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Kai Wang Boris Babenko and Serge Belongie. 2011. End-to-end scene text recognition. In ICCV.","DOI":"10.1109\/ICCV.2011.6126402"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Peng Wang Cheng Da and Cong Yao. 2022. Multi-granularity prediction for scene text recognition. In ECCV.","DOI":"10.1007\/978-3-031-19815-1_20"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"Wenhai Wang Enze Xie Xiang Li Deng-Ping Fan Kaitao Song Ding Liang Tong Lu Ping Luo and Ling Shao. 2021. Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In ICCV.","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Yuxin Wang Hongtao Xie Shancheng Fang Jing Wang Shenggao Zhu and Yongdong Zhang. 2021. From two to one: A new scene text recognizer with visual language modeling network. In ICCV.","DOI":"10.1109\/ICCV48922.2021.01393"},{"key":"e_1_3_2_1_44_1","first-page":"2404","article-title":"A two-level rectification attention network for scene text recognition","volume":"25","author":"Wu Lintai","year":"2022","unstructured":"Lintai Wu, Yong Xu, Junhui Hou, CL Philip Chen, and Cheng-Lin Liu. 2022. A two-level rectification attention network for scene text recognition. IEEE TMM, Vol. 25 (2022), 2404--2414.","journal-title":"IEEE TMM"},{"key":"e_1_3_2_1_45_1","volume-title":"CROP: zero-shot cross-lingual named entity recognition with multilingual labeled sequence translation. arXiv preprint arXiv:2210.07022","author":"Yang Jian","year":"2022","unstructured":"Jian Yang, Shaohan Huang, Shuming Ma, Yuwei Yin, Li Dong, Dongdong Zhang, Hongcheng Guo, Zhoujun Li, and Furu Wei. 2022. CROP: zero-shot cross-lingual named entity recognition with multilingual labeled sequence translation. arXiv preprint arXiv:2210.07022 (2022)."},{"key":"e_1_3_2_1_46_1","volume-title":"Ganlm: Encoder-decoder pre-training with an auxiliary discriminator. arXiv preprint arXiv:2212.10218","author":"Yang Jian","year":"2022","unstructured":"Jian Yang, Shuming Ma, Li Dong, Shaohan Huang, Haoyang Huang, Yuwei Yin, Dongdong Zhang, Liqun Yang, Furu Wei, and Zhoujun Li. 2022. Ganlm: Encoder-decoder pre-training with an auxiliary discriminator. arXiv preprint arXiv:2212.10218 (2022)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Jian Yang Shuming Ma Dongdong Zhang Shuangzhi Wu Zhoujun Li and Ming Zhou. 2020. Alternating language modeling for cross-lingual pre-training. In AAAI.","DOI":"10.1609\/aaai.v34i05.6480"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Jian Yang Yuwei Yin Shuming Ma Haoyang Huang Dongdong Zhang Zhoujun Li and Furu Wei. 2021. Multilingual agreement for multilingual neural machine translation. In ACL.","DOI":"10.18653\/v1\/2021.acl-short.31"},{"key":"e_1_3_2_1_49_1","unstructured":"Deli Yu Xuan Li Chengquan Zhang Tao Liu Junyu Han Jingtuo Liu and Errui Ding. 2020. Towards accurate scene text recognition with semantic reasoning networks. In CVPR."},{"key":"e_1_3_2_1_50_1","volume-title":"Benchmarking chinese text recognition: Datasets, baselines, and an empirical study. arXiv preprint arXiv:2112.15093","author":"Yu Haiyang","year":"2021","unstructured":"Haiyang Yu, Jingye Chen, Bin Li, Jianqi Ma, Mengnan Guan, Xixi Xu, Xiaocong Wang, Shaobo Qu, and Xiangyang Xue. 2021. Benchmarking chinese text recognition: Datasets, baselines, and an empirical study. arXiv preprint arXiv:2112.15093 (2021)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"Ziyin Zhang Ning Lu Minghui Liao Yongshuai Huang Cheng Li Min Wang and Wei Peng. 2024. Self-Distillation Regularized Connectionist Temporal Classification Loss for Text Recognition: A Simple Yet Effective Approach. In AAAI.","DOI":"10.1609\/aaai.v38i7.28575"},{"key":"e_1_3_2_1_52_1","volume-title":"Multi-stage pre-training enhanced by chatgpt for multi-scenario multi-domain dialogue summarization. arXiv preprint arXiv:2310.10285","author":"Zhou Weixiao","year":"2023","unstructured":"Weixiao Zhou, Gengyao Li, Xianfu Cheng, Xinnian Liang, Junnan Zhu, Feifei Zhai, and Zhoujun Li. 2023. Multi-stage pre-training enhanced by chatgpt for multi-scenario multi-domain dialogue summarization. arXiv preprint arXiv:2310.10285 (2023)."}],"event":{"name":"CIKM '24: The 33rd ACM International Conference on Information and Knowledge Management","location":"Boise ID USA","acronym":"CIKM '24","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 33rd ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627673.3679618","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3627673.3679618","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:58:23Z","timestamp":1750294703000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627673.3679618"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,21]]},"references-count":52,"alternative-id":["10.1145\/3627673.3679618","10.1145\/3627673"],"URL":"https:\/\/doi.org\/10.1145\/3627673.3679618","relation":{},"subject":[],"published":{"date-parts":[[2024,10,21]]},"assertion":[{"value":"2024-10-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}