{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,19]],"date-time":"2026-04-19T04:13:19Z","timestamp":1776571999808,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"111 project","award":["BP0719010"],"award-info":[{"award-number":["BP0719010"]}]},{"name":"STCSM","award":["22DZ2229005"],"award-info":[{"award-number":["22DZ2229005"]}]},{"name":"NSFC","award":["62171282"],"award-info":[{"award-number":["62171282"]}]},{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0102"],"award-info":[{"award-number":["2021SHZDZX0102"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612247","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"5764-5775","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["Relational Contrastive Learning for Scene Text Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-2660-1100","authenticated-orcid":false,"given":"Jinglei","family":"Zhang","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6761-5152","authenticated-orcid":false,"given":"Tiancheng","family":"Lin","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6508-4469","authenticated-orcid":false,"given":"Yi","family":"Xu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3636-9749","authenticated-orcid":false,"given":"Kai","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2523-074X","authenticated-orcid":false,"given":"Rui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01505"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00481"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00313"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"e_1_3_2_1_5_1","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume":"33","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020a. Unsupervised learning of visual features by contrasting cluster assignments. Advances in Neural Information Processing Systems, Vol. 33 (2020), 9912--9924.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume":"33","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020b. Unsupervised learning of visual features by contrasting cluster assignments. Advances in Neural Information Processing Systems, Vol. 33 (2020), 9912--9924.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","volume-title":"International conference on machine learning. PMLR, JMLR, 1269 LAW ST, SAN DIEGO, CA, UNITED STATES, 1597--1607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020b. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, JMLR, 1269 LAW ST, SAN DIEGO, CA, UNITED STATES, 1597--1607."},{"key":"e_1_3_2_1_8_1","volume-title":"Improved Baselines with Momentum Contrastive Learning. arxiv","author":"Chen Xinlei","year":"2003","unstructured":"Xinlei Chen, Haoqi Fan, Ross Girshick, and Kaiming He. 2020a. Improved Baselines with Momentum Contrastive Learning. arxiv: 2003.04297 [cs.CV]"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.543"},{"key":"e_1_3_2_1_11_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. arxiv","author":"Dosovitskiy Alexey","year":"2010","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. arxiv: 2010.11929 [cs.CV]"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_2_1_15_1","volume-title":"Zhaohan Guo, Mohammad Gheshlaghi Azar, et al.","author":"Grill Jean-Bastien","year":"2020","unstructured":"Jean-Bastien Grill, Florian Strub, Florent Altch\u00e9, Corentin Tallec, Pierre Richemond, Elena Buchatskaya, Carl Doersch, Bernardo Avila Pires, Zhaohan Guo, Mohammad Gheshlaghi Azar, et al. 2020. Bootstrap your own latent-a new approach to self-supervised learning. Advances in neural information processing systems, Vol. 33 (2020), 21271--21284."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.254"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/2968618.2968725"},{"key":"e_1_3_2_1_20_1","unstructured":"Max Jaderberg Karen Simonyan Andrea Vedaldi and Andrew Zisserman. 2014. Synthetic Data and Artificial Neural Networks for Natural Scene Text Recognition. arxiv: 1406.2227 [cs.CV]"},{"key":"e_1_3_2_1_21_1","volume-title":"ICDAR 2015 competition on robust reading. In 2015 13th international conference on document analysis and recognition (ICDAR). IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1156--1160","author":"Karatzas Dimosthenis","year":"2015","unstructured":"Dimosthenis Karatzas, Lluis Gomez-Bigorda, Anguelos Nicolaou, Suman Ghosh, Andrew Bagdanov, Masakazu Iwamura, Jiri Matas, Lukas Neumann, Vijay Ramaseshan Chandrasekhar, Shijian Lu, et al. 2015. ICDAR 2015 competition on robust reading. In 2015 13th international conference on document analysis and recognition (ICDAR). IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1156--1160."},{"key":"e_1_3_2_1_22_1","volume-title":"ICDAR 2013 robust reading competition. In 2013 12th international conference on document analysis and recognition. IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1484--1493","author":"Karatzas Dimosthenis","year":"2013","unstructured":"Dimosthenis Karatzas, Faisal Shafait, Seiichi Uchida, Masakazu Iwamura, Lluis Gomez i Bigorda, Sergi Robles Mestre, Joan Mas, David Fernandez Mota, Jon Almazan Almazan, and Lluis Pere De Las Heras. 2013. ICDAR 2013 robust reading competition. In 2013 12th international conference on document analysis and recognition. IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1484--1493."},{"key":"e_1_3_2_1_23_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2017","unstructured":"Diederik P. Kingma and Jimmy Ba. 2017. Adam: A Method for Stochastic Optimization. arxiv: 1412.6980 [cs.LG]"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.117"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00281"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00281"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Hao Liu Bin Wang Zhimin Bao Mobai Xue Sheng Kang Deqiang Jiang Yinsong Liu and Bo Ren. 2022. Perceiving Stroke-Semantic Context: Hierarchical Contrastive Learning for Robust Scene Text Recognition. In AAAI. AAAI 2275 E BAYSHORE RD STE 160 PALO ALTO CA 94303 USA 1702--1710.","DOI":"10.1609\/aaai.v36i2.20062"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-004-0134-3"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00111"},{"key":"e_1_3_2_1_30_1","unstructured":"Pengyuan Lyu Chengquan Zhang Shanshan Liu Meina Qiao Yangliu Xu Liang Wu Kun Yao Junyu Han Errui Ding and Jingdong Wang. 2022. MaskOCR: Text Recognition with Masked Encoder-Decoder Pretraining. arxiv: 2206.00311 [cs.CV]"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s100320200071"},{"key":"e_1_3_2_1_32_1","first-page":"1","article-title":"Scene text recognition using higher order language priors. In BMVC-British machine vision conference. BMVA, B M V A PRESS, 49A ELMSIDE ONSLOW VILLAGE, GUILDFORD, SURREY GU2 5SX","volume":"127","author":"Mishra Anand","year":"2012","unstructured":"Anand Mishra, Karteek Alahari, and CV Jawahar. 2012. Scene text recognition using higher order language priors. In BMVC-British machine vision conference. BMVA, B M V A PRESS, 49A ELMSIDE ONSLOW VILLAGE, GUILDFORD, SURREY GU2 5SX, ENGLAND, 127.1--127.11.","journal-title":"ENGLAND"},{"key":"e_1_3_2_1_33_1","volume-title":"Representation Learning via Invariant Causal Mechanisms. arxiv","author":"Mitrovic Jovana","year":"2010","unstructured":"Jovana Mitrovic, Brian McWilliams, Jacob Walker, Lars Buesing, and Charles Blundell. 2020. Representation Learning via Invariant Causal Mechanisms. arxiv: 2010.07922 [cs.LG]"},{"key":"e_1_3_2_1_34_1","first-page":"4003","article-title":"Self-supervised relational reasoning for representation learning","volume":"33","author":"Patacchiola Massimiliano","year":"2020","unstructured":"Massimiliano Patacchiola and Amos J Storkey. 2020. Self-supervised relational reasoning for representation learning. Advances in Neural Information Processing Systems, Vol. 33 (2020), 4003--4014.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.76"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01354"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2014.07.008"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Rico Sennrich Barry Haddow and Alexandra Birch. 2016. Neural Machine Translation of Rare Words with Subword Units. arxiv: 1508.07909 [cs.CL]","DOI":"10.18653\/v1\/P16-1162"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.452"},{"key":"e_1_3_2_1_40_1","volume-title":"Super-convergence: Very fast training of neural networks using large learning rates. In Artificial intelligence and machine learning for multi-domain operations applications","author":"Smith Leslie N","year":"2019","unstructured":"Leslie N Smith and Nicholay Topin. 2019. Super-convergence: Very fast training of neural networks using large learning rates. In Artificial intelligence and machine learning for multi-domain operations applications, Vol. 11006. SPIE, SPIE-INT SOC OPTICAL ENGINEERING, 1000 20TH ST, PO BOX 10, BELLINGHAM, WA 98227-0010 USA, 369--386."},{"key":"e_1_3_2_1_41_1","volume-title":"Representation Learning with Contrastive Predictive Coding. arxiv","author":"van den Oord Aaron","year":"1807","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2019. Representation Learning with Contrastive Predictive Coding. arxiv: 1807.03748 [cs.LG]"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01144"},{"key":"e_1_3_2_1_43_1","volume-title":"2011 International conference on computer vision. IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1457--1464","author":"Wang Kai","year":"2011","unstructured":"Kai Wang, Boris Babenko, and Serge Belongie. 2011. End-to-end scene text recognition. In 2011 International conference on computer vision. IEEE, IEEE, 345 E 47TH ST, NEW YORK, NY 10017 USA, 1457--1464."},{"key":"e_1_3_2_1_44_1","first-page":"5549","article-title":"Contrastive learning with stronger augmentations","volume":"45","author":"Wang Xiao","year":"2022","unstructured":"Xiao Wang and Guo-Jun Qi. 2022. Contrastive learning with stronger augmentations. IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 45 (2022), 5549--5560.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01393"},{"key":"e_1_3_2_1_46_1","volume-title":"CO2: Consistent Contrast for Unsupervised Visual Representation Learning. arxiv","author":"Wei Chen","year":"2010","unstructured":"Chen Wei, Huiyu Wang, Wei Shen, and Alan Yuille. 2020. CO2: Consistent Contrast for Unsupervised Visual Representation Learning. arxiv: 2010.02217 [cs.CV]"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00393"},{"key":"e_1_3_2_1_48_1","volume-title":"Tel Aviv","author":"Xie Xudong","year":"2022","unstructured":"Xudong Xie, Ling Fu, Zhifei Zhang, Zhaowen Wang, and Xiang Bai. 2022a. Toward Understanding WordArt: Corner-Guided Transformer for Scene Text Recognition. In Computer Vision-ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXVIII. Springer, SPRINGER INTERNATIONAL PUBLISHING AG, GEWERBESTRASSE 11, CHAM, CH-6330, SWITZERLAND, 303--321."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547784"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01213"},{"key":"e_1_3_2_1_52_1","unstructured":"Haiyang Yu Jingye Chen Bin Li Jianqi Ma Mengnan Guan Xixi Xu Xiaocong Wang Shaobo Qu and Xiangyang Xue. 2022. Benchmarking Chinese Text Recognition: Datasets Baselines and an Empirical Study. arxiv: 2112.15093 [cs.CV]"},{"key":"e_1_3_2_1_53_1","volume-title":"ADADELTA: An Adaptive Learning Rate Method. arxiv: 1212.5701 [cs.LG]","author":"Zeiler Matthew D.","year":"2012","unstructured":"Matthew D. Zeiler. 2012. ADADELTA: An Adaptive Learning Rate Method. arxiv: 1212.5701 [cs.LG]"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20245"},{"key":"e_1_3_2_1_55_1","first-page":"2543","article-title":"Ressl: Relational self-supervised learning with weak augmentation","volume":"34","author":"Zheng Mingkai","year":"2021","unstructured":"Mingkai Zheng, Shan You, Fei Wang, Chen Qian, Changshui Zhang, Xiaogang Wang, and Chang Xu. 2021. Ressl: Relational self-supervised learning with weak augmentation. Advances in Neural Information Processing Systems, Vol. 34 (2021), 2543--2555.","journal-title":"Advances in Neural Information Processing Systems"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612247","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612247","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:58:48Z","timestamp":1755820728000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612247"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":55,"alternative-id":["10.1145\/3581783.3612247","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612247","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}