{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T04:55:23Z","timestamp":1780635323596,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["92370119, 62436009, 62276258 and 62376113"],"award-info":[{"award-number":["92370119, 62436009, 62276258 and 62376113"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763913","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:30:41Z","timestamp":1765211441000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["DvD: Unleashing a Generative Paradigm for Document Dewarping via Coordinates-based Diffusion Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8783-0326","authenticated-orcid":false,"given":"Weiguang","family":"Zhang","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong-Liverpool University, Suzhou, China and University of Liverpool, Liverpool, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9808-8823","authenticated-orcid":false,"given":"Huangcheng","family":"Lu","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong-Liverpool University, Suzhou, China and University of Liverpool, Liverpool, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8842-4187","authenticated-orcid":false,"given":"Maizhen","family":"Ning","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong-Liverpool University, Suzhou, China and University of Liverpool, Liverpool, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6267-0366","authenticated-orcid":false,"given":"Xiaowei","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Liverpool, Liverpool, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0707-8076","authenticated-orcid":false,"given":"Wei","family":"Wang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong-Liverpool University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3034-9639","authenticated-orcid":false,"given":"Kaizhu","family":"Huang","sequence":"additional","affiliation":[{"name":"Duke Kunshan University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0918-4606","authenticated-orcid":false,"given":"Qiufeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong-Liverpool University, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et\u00a0al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2001.937649"},{"key":"e_1_3_3_2_4_1","first-page":"228","volume-title":"International Conference on Computer Vision(ICCV)","author":"Cao Huaigu","year":"2003","unstructured":"Huaigu Cao, Xiaoqing Ding, and Changsong Liu. 2003a. A cylindrical surface model to rectify the bound document image. In International Conference on Computer Vision(ICCV). 228\u2013233."},{"key":"e_1_3_3_2_5_1","first-page":"71","volume-title":"International Conference on Document Analysis and Recognition(ICDAR)","author":"Cao Huaigu","year":"2003","unstructured":"Huaigu Cao, Xiaoqing Ding, and Changsong Liu. 2003b. Rectifying the bound document image captured by the camera: A model based approach. In International Conference on Document Analysis and Recognition(ICDAR). IEEE, 71\u201375."},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00022"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00423"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"crossref","unstructured":"Hao Feng Qi Liu Hao Liu Jingqun Tang Wengang Zhou Houqiang Li and Can Huang. 2024. DocPedia: unleashing the power of large multimodal model in the frequency domain for versatile document understanding. Science China Information Sciences 67 12 (Dec. 2024) 220106.","DOI":"10.1007\/s11432-024-4250-y"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475388"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"crossref","unstructured":"Hao Feng Wengang Zhou Jiajun Deng Qi Tian and Houqiang Li. 2025. DocScanner: Robust document image rectification with progressive learning. International Journal of Computer Vision(IJCV) (2025).","DOI":"10.1007\/s11263-025-02431-5"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_27"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Felix Hertlein Alexander Naumann and Patrick Philipp. 2023. Inv3D: a high-resolution 3D invoice dataset for template-guided single-image document unwarping. International Journal on Document Analysis and Recognition(IJDAR) (2023) 1\u201312.","DOI":"10.1007\/s10032-023-00434-x"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00563"},{"key":"e_1_3_3_2_14_1","first-page":"6840","volume-title":"Advances in Neural Information Processing Systems(NIPS)","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. In Advances in Neural Information Processing Systems(NIPS) , H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.F. Balcan, and H.\u00a0Lin (Eds.), Vol.\u00a033. Curran Associates, Inc., 6840\u20136851."},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00450"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.1993.395633"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","unstructured":"Ilia Karmanov Amala\u00a0Sanjay Deshmukh Lukas Voegtle Philipp Fischer Kateryna Chumachenko Timo Roman Jarno Sepp\u00e4nen Jupinder Parmar Joseph Jennings Andrew Tao and Karan Sapra. 2025. \u00c9clair \u2013 Extracting Content and Layout with Integrated Reading Order for Documents. 10.48550\/arXiv.2502.04223arXiv:https:\/\/arXiv.org\/abs\/2502.04223 [cs].","DOI":"10.48550\/arXiv.2502.04223"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"crossref","unstructured":"Beom\u00a0Su Kim Hyung\u00a0Il Koo and Nam\u00a0Ik Cho. 2015. Document dewarping via text-line based optimization. Pattern Recognition(PR) 48 11 (2015) 3600\u20133614.","DOI":"10.1016\/j.patcog.2015.04.026"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"crossref","unstructured":"Hyung\u00a0Il Koo Jinho Kim and Nam\u00a0Ik Cho. 2009. Composition of a dewarped and enhanced document image from two view images. IEEE Transactions on Image Processing(TIP) 18 7 (2009) 1551\u20131562.","DOI":"10.1109\/TIP.2009.2019301"},{"key":"e_1_3_3_2_20_1","unstructured":"Vladimir\u00a0I Levenshtein et\u00a0al. 1966. Binary codes capable of correcting deletions insertions and reversals. Soviet Physics Doklady 10 8 (1966) 707\u2013710."},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01793"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"crossref","unstructured":"Pu Li Weize Quan Jianwei Guo and Dong-Ming Yan. 2023a. Layout-Aware Single-Image Document Flattening. ACM Transactions on Graphics(TOG) 43 1 (2023).","DOI":"10.1145\/3627818"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"crossref","unstructured":"Xiaoyu Li Bo Zhang Jing Liao and Pedro\u00a0V Sander. 2019. Document rectification and illumination correction using a patch-based CNN. ACM Transactions on Graphics(TOG) 38 6 (2019) 1\u201311.","DOI":"10.1145\/3355089.3356563"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01399"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02527"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"crossref","unstructured":"Jian Liang Daniel DeMenthon and David Doermann. 2008. Geometric rectification of camera-captured document images. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 30 4 (2008) 591\u2013605.","DOI":"10.1109\/TPAMI.2007.70724"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"crossref","unstructured":"Ce Liu Jenny Yuen and Antonio Torralba. 2010. Sift flow: Dense correspondence across scenes and its applications. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 33 5 (2010) 978\u2013994.","DOI":"10.1109\/TPAMI.2010.147"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01813"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3528233.3530756"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00494"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58610-2_13"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.447"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01270-0_11"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.497"},{"key":"e_1_3_3_2_35_1","unstructured":"Jisu Nam Gyuseong Lee Sunwoo Kim Hyeonsu Kim Hyoungwon Cho Seyeon Kim and Seungryong Kim. 2024. DiffMatch: Diffusion Model for Dense Matching. (2024)."},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","unstructured":"Ahmed Nassar Andres Marafioti Matteo Omenetti Maksym Lysak Nikolaos Livathinos Christoph Auer Lucas Morin Rafael Teixeira\u00a0de Lima Yusik Kim A.\u00a0Said Gurbuz Michele Dolfi Miquel Farr\u00e9 and Peter W.\u00a0J. Staar. 2025. SmolDocling: An ultra-compact vision-language model for end-to-end multi-modal document conversion. 10.48550\/arXiv.2503.11576arXiv:https:\/\/arXiv.org\/abs\/2503.11576 [cs].","DOI":"10.48550\/arXiv.2503.11576"},{"key":"e_1_3_3_2_37_1","series-title":"Proceedings of Machine Learning Research","first-page":"8162","volume-title":"International Conference on Machine Learning(ICML)","volume":"139","author":"Nichol Alexander\u00a0Quinn","year":"2021","unstructured":"Alexander\u00a0Quinn Nichol and Prafulla Dhariwal. 2021. Improved Denoising Diffusion Probabilistic Models. In International Conference on Machine Learning(ICML)(Proceedings of Machine Learning Research, Vol.\u00a0139), Marina Meila and Tong Zhang (Eds.). PMLR, 8162\u20138171."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"crossref","unstructured":"Xuebin Qin Zichen Zhang Chenyang Huang Masood Dehghan Osmar\u00a0R. Zaiane and Martin Jagersand. 2020. U2-Net: Going deeper with nested U-structure for salient object detection. Pattern Recognition(PR) 106 (2020) 107404.","DOI":"10.1016\/j.patcog.2020.107404"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","unstructured":"Rahul Ravishankar Zeeshan Patel Jathushan Rajasegaran and Jitendra Malik. 2024. Scaling Properties of Diffusion Models for Perceptual Tasks. 10.48550\/arXiv.2411.08034arXiv:https:\/\/arXiv.org\/abs\/2411.08034 [cs].","DOI":"10.48550\/arXiv.2411.08034"},{"key":"e_1_3_3_2_41_1","series-title":"International Conference on Machine Learning(ICML)","first-page":"1278","volume-title":"Proceedings of the 31st International Conference on Machine Learning","volume":"32","author":"Rezende Danilo\u00a0Jimenez","year":"2014","unstructured":"Danilo\u00a0Jimenez Rezende, Shakir Mohamed, and Daan Wierstra. 2014. Stochastic Backpropagation and Approximate Inference in Deep Generative Models. In Proceedings of the 31st International Conference on Machine Learning(International Conference on Machine Learning(ICML), Vol.\u00a032), Eric\u00a0P. Xing and Tony Jebara (Eds.). PMLR, Bejing, China, 1278\u20131286."},{"key":"e_1_3_3_2_42_1","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. arxiv:https:\/\/arXiv.org\/abs\/2112.10752\u00a0[cs.CV]"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-70546-5_11"},{"key":"e_1_3_3_2_45_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1409.1556 (2014)."},{"key":"e_1_3_3_2_46_1","volume-title":"Advances in Neural Information Processing Systems(NIPS)","author":"Song Yang","year":"2019","unstructured":"Yang Song and Stefano Ermon. 2019. Generative Modeling by Estimating Gradients of the Data Distribution. In Advances in Neural Information Processing Systems(NIPS) , H.\u00a0Wallach, H.\u00a0Larochelle, A.\u00a0Beygelzimer, F.\u00a0d'Alch\u00e9-Buc, E.\u00a0Fox, and R.\u00a0Garnett (Eds.), Vol.\u00a032. Curran Associates, Inc."},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"crossref","unstructured":"Chew\u00a0Lim Tan Li Zhang Zheng Zhang and Tao Xia. 2005. Restoring warped document images through 3D shape modeling. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 28 2 (2005) 195\u2013208.","DOI":"10.1109\/TPAMI.2006.40"},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995540"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2007.383251"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/1030397.1030434"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618174"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"crossref","unstructured":"Toshikazu Wada Hiroyuki Ukida and Takashi Matsuyama. 1997. Shape from shading with interreflections under a proximal light source: Distortion-free copying of an unfolded book. International Journal of Computer Vision(IJCV) 24 2 (1997) 125\u2013135.","DOI":"10.1023\/A:1007906904009"},{"key":"e_1_3_3_2_53_1","series-title":"(ICML\u201924)","volume-title":"International Conference on Machine Learning(ICML)","author":"Wan Xingyu","year":"2024","unstructured":"Xingyu Wan, Chengquan Zhang, Pengyuan Lyu, Sen Fan, Zihan Ni, Kun Yao, Errui Ding, and Jingdong Wang. 2024. Towards unified multi-granularity text detection with interactive attention. In International Conference on Machine Learning(ICML) (Vienna, Austria) (ICML\u201924). JMLR.org, Article 2046, 14\u00a0pages."},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACSSC.2003.1292216"},{"key":"e_1_3_3_2_55_1","unstructured":"Haoran Wei Chenglong Liu Jinyue Chen Jia Wang Lingyu Kong Yanming Xu Zheng Ge Liang Zhao Jianjian Sun Yuang Peng et\u00a0al. 2024. General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.01704 (2024)."},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-57058-3_10"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86549-8_30"},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00453"},{"key":"e_1_3_3_2_59_1","unstructured":"Zhiyuan Yan Junyan Ye Weijia Li Zilong Huang Shenghai Yuan Xiangyang He Kaiqing Lin Jun He Conghui He and Li Yuan. 2025. GPT-ImgEval: A Comprehensive Benchmark for Diagnosing GPT4o in Image Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.02782 (2025)."},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"crossref","unstructured":"Shaodi You Yasuyuki Matsushita Sudipta Sinha Yusuke Bou and Katsushi Ikeuchi. 2017a. Multiview rectification of folded documents. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 40 2 (2017) 505\u2013511.","DOI":"10.1109\/TPAMI.2017.2675980"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"crossref","unstructured":"Shaodi You Yasuyuki Matsushita Sudipta Sinha Yusuke Bou and Katsushi Ikeuchi. 2017b. Multiview rectification of folded documents. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 40 2 (2017) 505\u2013511.","DOI":"10.1109\/TPAMI.2017.2675980"},{"key":"e_1_3_3_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00072"},{"key":"e_1_3_3_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548214"},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01482"},{"key":"e_1_3_3_2_65_1","doi-asserted-by":"crossref","unstructured":"Li Zhang Yu Zhang and Chew Tan. 2008. An improved physically-based method for geometric restoration of distorted document images. IEEE Transactions on Pattern Analysis and Machine Intelligence(TPAMI) 30 4 (2008) 728\u2013734.","DOI":"10.1109\/TPAMI.2007.70831"},{"key":"e_1_3_3_2_66_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-70546-5_20"},{"key":"e_1_3_3_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681548"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-78119-3_3"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763913","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:30:44Z","timestamp":1765251044000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763913"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":67,"alternative-id":["10.1145\/3757377.3763913","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763913","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}