{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T14:55:29Z","timestamp":1778079329549,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681655","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"9847-9855","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["RDLNet: A Novel and Accurate Real-world Document Localization Method"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8830-8250","authenticated-orcid":false,"given":"Yaqiang","family":"Wu","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University &amp; Lenovo Research, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-8644-6975","authenticated-orcid":false,"given":"Zhen","family":"Xu","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1715-115X","authenticated-orcid":false,"given":"Yong","family":"Duan","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7498-9409","authenticated-orcid":false,"given":"Yanlai","family":"Wu","sequence":"additional","affiliation":[{"name":"Chongqing Jiaotong University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8436-4754","authenticated-orcid":false,"given":"Qinghua","family":"Zheng","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5625-8836","authenticated-orcid":false,"given":"Hui","family":"Li","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3636-0099","authenticated-orcid":false,"given":"Xiaochen","family":"Hu","sequence":"additional","affiliation":[{"name":"Lenovo Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5456-0957","authenticated-orcid":false,"given":"Lianwen","family":"Jin","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"818","article-title":"MIDV-500: a dataset for identity document analysis and recognition on mobile devices in video stream","volume":"43","author":"Arlazarov Vladimir Viktorovich","year":"2019","unstructured":"Vladimir Viktorovich Arlazarov, Konstantin Bulatovich Bulatov, Timofey Sergeevich Chernov, and Vladimir Lvovich Arlazarov. 2019. MIDV-500: a dataset for identity document analysis and recognition on mobile devices in video stream. CoOpt, Vol. 43, 5 (2019), 818--824.","journal-title":"CoOpt"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1117\/12.2558438"},{"key":"e_1_3_2_1_3_1","first-page":"252","article-title":"MIDV-2020: a comprehensive benchmark dataset for identity document analysis","volume":"46","author":"Bulatovich Bulatov Konstantin","year":"2022","unstructured":"Bulatov Konstantin Bulatovich, Emelianova Ekaterina Vladimirovna, Tropin Daniil Vyacheslavovich, Skoryukina Natalya Sergeevna, Chernyshova Yulia Sergeevna, Ming Zuheng, Burie Jean-Christophe, and Luqman Muhammad Muzzamil. 2022. MIDV-2020: a comprehensive benchmark dataset for identity document analysis. CoOpt, Vol. 46, 2 (2022), 252--270.","journal-title":"CoOpt"},{"key":"e_1_3_2_1_4_1","volume-title":"ICDAR2015 competition on smartphone document capture and OCR (SmartDoc). In 2015 13th International Conference on Document Analysis and Recognition (ICDAR). IEEE, 1161--1165","author":"Burie Jean-Christophe","year":"2015","unstructured":"Jean-Christophe Burie, Joseph Chazalon, Micka\u00ebl Coustaty, S\u00e9bastien Eskenazi, Muhammad Muzzamil Luqman, Maroua Mehri, Nibal Nayef, Jean-Marc Ogier, Sophea Prum, and Marccal Rusi nol. 2015. ICDAR2015 competition on smartphone document capture and OCR (SmartDoc). In 2015 13th International Conference on Document Analysis and Recognition (ICDAR). IEEE, 1161--1165."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Alejandra Castelblanco Jesus Solano Christian Lopez Esteban Rivera and Mart\u00edn Ochoa. 2020. Machine Learning Techniques for Identity Document Verification in Uncontrolled Environments: A Case Study. (2020).","DOI":"10.1007\/978-3-030-49076-8_26"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1049\/iet-ipr.2020.0532"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9206711"},{"key":"e_1_3_2_1_9_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2008.300"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.26"},{"key":"e_1_3_2_1_13_1","volume-title":"Data Efficient Training of a U-Net Based Architecture for Structured Documents Localization. arXiv preprint arXiv:2310.00937","author":"Kabeshova Anastasiia","year":"2023","unstructured":"Anastasiia Kabeshova, Guillaume Betmont, Julien Lerouge, Evgeny Stepankevich, and Alexis Berg\u00e8s. 2023. Data Efficient Training of a U-Net Based Architecture for Structured Documents Localization. arXiv preprint arXiv:2310.00937 (2023)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings, Part V 13","author":"Kr\u00e4henb\u00fchl Philipp","year":"2014","unstructured":"Philipp Kr\u00e4henb\u00fchl and Vladlen Koltun. 2014. Geodesic object proposals. In Computer Vision--ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6--12, 2014, Proceedings, Part V 13. Springer, 725--739."},{"key":"e_1_3_2_1_16_1","volume-title":"Oblivious Document Capture and Real-Time Retrieval. In International Workshop on Camera Based Document Analysis and Recognition (CBDAR).","author":"Lampert Christoph H","year":"2005","unstructured":"Christoph H Lampert, Tim Braun, Adrian Ulges, Daniel Keysers, and Thomas M Breuel. 2005. Oblivious Document Capture and Real-Time Retrieval. In International Workshop on Camera Based Document Analysis and Recognition (CBDAR)."},{"key":"e_1_3_2_1_17_1","volume-title":"Shilong Liu, Lei Zhang, Lionel M. Ni, and Heung-Yeung Shum.","author":"Li Feng","year":"2022","unstructured":"Feng Li, Hao Zhang, Huaizhe xu, Shilong Liu, Lei Zhang, Lionel M. Ni, and Heung-Yeung Shum. 2022. Mask DINO: Towards A Unified Transformer-based Framework for Object Detection and Segmentation. arxiv: 2206.02777 [cs.CV]"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"e_1_3_2_1_19_1","volume-title":"7th International Workshop, DAS 2006, Nelson, New Zealand, February 13--15, 2006, Proceedings.","author":"Lu Shijian","year":"2006","unstructured":"Shijian Lu and Chew Lim Tan. 2006. The Restoration of Camera Documents Through Image Segmentation. In Document Analysis Systems VII, 7th International Workshop, DAS 2006, Nelson, New Zealand, February 13--15, 2006, Proceedings."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCIAS.2006.295374"},{"key":"e_1_3_2_1_21_1","volume-title":"V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). Ieee, 565--571.","author":"Milletari Fausto","year":"2016","unstructured":"Fausto Milletari, Nassir Navab, and Seyed-Ahmad Ahmadi. 2016. V-net: Fully convolutional neural networks for volumetric medical image segmentation. In 2016 fourth international conference on 3D vision (3DV). Ieee, 565--571."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.395"},{"key":"e_1_3_2_1_23_1","first-page":"40","article-title":"Thresholding based on variance and intensity contrast","volume":"2","author":"Qiao Y","year":"2007","unstructured":"Y Qiao, Q. M. Hu, G. Y. Qian, S. H. Luo, and W. L. Nowinski. 2007. Thresholding based on variance and intensity contrast. Pattern Recognition: The Journal of the Pattern Recognition Society 2 (2007), 40.","journal-title":"Pattern Recognition: The Journal of the Pattern Recognition Society"},{"key":"e_1_3_2_1_24_1","volume-title":"U-net: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention--MICCAI 2015: 18th international conference","author":"Ronneberger Olaf","year":"2015","unstructured":"Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention--MICCAI 2015: 18th international conference, Munich, Germany, October 5--9, 2015, proceedings, part III 18. Springer, 234--241."},{"key":"e_1_3_2_1_25_1","volume-title":"Document Localization Algorithms Based on Feature Points and Straight Lines. In International Conference on Machine Vision.","author":"Skoryukina Natalya","year":"2018","unstructured":"Natalya Skoryukina, Julia Shemiakina, Vladimir L. Arlazarov, and Igor Faradjev. 2018. Document Localization Algorithms Based on Feature Points and Straight Lines. In International Conference on Machine Vision."},{"key":"e_1_3_2_1_26_1","volume-title":"2nd International Workshop on Camera-Based Document Analysis and Recognition","author":"Stamatopoulos N","year":"2007","unstructured":"N Stamatopoulos, B Gatos, and A Kesidis. 2007. Automatic borders detection of camera document images. In 2nd International Workshop on Camera-Based Document Analysis and Recognition, Curitiba, Brazil. 71--78."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.446"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00359"},{"key":"e_1_3_2_1_29_1","volume-title":"LDRNet: Enabling Real-time Document Localization on Mobile Devices. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 618--629","author":"Wu Han","year":"2022","unstructured":"Han Wu, Holland Qian, Huaming Wu, and Aad van Moorsel. 2022. LDRNet: Enabling Real-time Document Localization on Mobile Devices. In Joint European Conference on Machine Learning and Knowledge Discovery in Databases. Springer, 618--629."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2024.3371348"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.164"},{"key":"e_1_3_2_1_32_1","volume-title":"EfficientSAM: Leveraged Masked Image Pretraining for Efficient Segment Anything. arXiv:2312.00863","author":"Xiong Yunyang","year":"2023","unstructured":"Yunyang Xiong, Bala Varadarajan, Lemeng Wu, Xiaoyu Xiang, Fanyi Xiao, Chenchen Zhu, Xiaoliang Dai, Dilin Wang, Fei Sun, Forrest Iandola, Raghuraman Krishnamoorthi, and Vikas Chandra. 2023. EfficientSAM: Leveraged Masked Image Pretraining for Efficient Segment Anything. arXiv:2312.00863 (2023)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2554550"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01733"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-019-00341-0"},{"key":"e_1_3_2_1_36_1","volume-title":"Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159","author":"Zhu Xizhou","year":"2020","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681655","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681655","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:49Z","timestamp":1750295869000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681655"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":36,"alternative-id":["10.1145\/3664647.3681655","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681655","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}