{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,4]],"date-time":"2026-04-04T21:44:36Z","timestamp":1775339076317,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,15]],"date-time":"2023-12-15T00:00:00Z","timestamp":1702598400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,15]]},"DOI":"10.1145\/3627631.3627639","type":"proceedings-article","created":{"date-parts":[[2024,1,31]],"date-time":"2024-01-31T12:08:32Z","timestamp":1706702912000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["TransDocUNet: A Transformer-based UNet Architecture for Degraded Document Image Binarization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9906-0609","authenticated-orcid":false,"given":"Risab","family":"Biswas","sequence":"first","affiliation":[{"name":"Artificial Intelligence Group, Optiks Innovations Pvt. Ltd. (P360), India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7198-3004","authenticated-orcid":false,"given":"Soumik","family":"Sarkhel","sequence":"additional","affiliation":[{"name":"Artificial Intelligence Group, Optiks Innovations Pvt. Ltd. (P360), India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6580-3977","authenticated-orcid":false,"given":"Swalpa Kumar","family":"Roy","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Alipurduar Government Engineering and Management College, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5426-2618","authenticated-orcid":false,"given":"Umapada","family":"Pal","sequence":"additional","affiliation":[{"name":"Computer Vision and Pattern Recognition Unit, Indian Statistical Institute, Kolkata, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,1,31]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.228"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2809544.2809561"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"e_1_3_2_1_4_1","volume-title":"U-Net-bin: hacking the document image binarization contest. 43, 5","author":"Bezmaternykh Pavel\u00a0Vladimirovich","year":"2019","unstructured":"Pavel\u00a0Vladimirovich Bezmaternykh, Dmitrii\u00a0Alexeevich Ilin, and Dmitry\u00a0Petrovich Nikolaev. 2019. U-Net-bin: hacking the document image binarization contest. 43, 5 (2019), 825\u2013832."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2019.8803348"},{"key":"e_1_3_2_1_6_1","unstructured":"Risab Biswas. 2023. Polyp-SAM++: Can A Text Guided SAM Perform Better for Polyp Segmentation?arxiv:2308.06623\u00a0[eess.IV]"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Risab Biswas Avirup Basu Abhishek Nandy Arkaprova Deb Roshni Chowdhury and Debashree Chanda. 2020. Identification of Pathological Disease in Plants using Deep Neural Networks-Powered by Intel\u00ae Distribution of OpenVINO\u2122 Toolkit. In 2020 Indo\u2013Taiwan 2nd International Conference on Computing Analytics and Networks (Indo-Taiwan ICAN). IEEE 45\u201348.","DOI":"10.1109\/Indo-TaiwanICAN48429.2020.9181339"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","unstructured":"Risab Biswas Avirup Basu Abhishek Nandy Arkaprova Deb Kazi Haque and Debashree Chanda. 2020. Drug Discovery and Drug Identification using AI. In 2020 Indo \u2013 Taiwan 2nd International Conference on Computing Analytics and Networks (Indo-Taiwan ICAN). 49\u201351. https:\/\/doi.org\/10.1109\/Indo-TaiwanICAN48429.2020.9181309","DOI":"10.1109\/Indo-TaiwanICAN48429.2020.9181309"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","unstructured":"Risab Biswas Swalpa\u00a0Kumar Roy and Umapada Pal. 2020. Effective Document Image Enhancement Using tokens-to-token Transformer Network. (2020). https:\/\/doi.org\/10.2139\/ssrn.4354038","DOI":"10.2139\/ssrn.4354038"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1080\/2151237X.2007.10129236"},{"key":"e_1_3_2_1_11_1","unstructured":"Jieneng Chen Yongyi Lu Qihang Yu Xiangde Luo Ehsan Adeli Yan Wang Le Lu Alan\u00a0L. Yuille and Yuyin Zhou. 2021. TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation. arxiv:2102.04306\u00a0[cs.CV]"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3062904"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571600.3571641"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571600.3571641"},{"key":"e_1_3_2_1_15_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2009.246"},{"key":"e_1_3_2_1_17_1","volume-title":"Adaptive degraded document image binarization. Pattern recognition 39, 3","author":"Gatos Basilios","year":"2006","unstructured":"Basilios Gatos, Ioannis Pratikakis, and Stavros\u00a0J Perantonis. 2006. Adaptive degraded document image binarization. Pattern recognition 39, 3 (2006), 317\u2013327."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.amc.2019.01.021"},{"key":"e_1_3_2_1_19_1","volume-title":"DeepOtsu: Document enhancement and binarization using iterative deep learning. Pattern recognition 91","author":"He Sheng","year":"2019","unstructured":"Sheng He and Lambert Schomaker. 2019. DeepOtsu: Document enhancement and binarization using iterative deep learning. Pattern recognition 91 (2019), 379\u2013390."},{"key":"e_1_3_2_1_20_1","volume-title":"Denoising Diffusion Probabilistic Models. arXiv preprint arxiv:2006.11239","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. arXiv preprint arxiv:2006.11239 (2020)."},{"key":"e_1_3_2_1_21_1","volume-title":"Long short-term memory. Neural computation 9, 8","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation 9, 8 (1997), 1735\u20131780."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2011.11"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108370"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107577"},{"key":"e_1_3_2_1_25_1","unstructured":"Bahjat Kawar Michael Elad Stefano Ermon and Jiaming Song. 2022. Denoising Diffusion Restoration Models. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_26_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma P","year":"2014","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412442"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86337-1_36"},{"key":"e_1_3_2_1_29_1","volume-title":"An introduction to digital image processing","author":"Niblack Wayne","unstructured":"Wayne Niblack. 1985. An introduction to digital image processing. Strandberg Publishing Company."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFHR.2014.141"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1979.4310076"},{"key":"e_1_3_2_1_32_1","volume-title":"International Work-Conference on Artificial Neural Networks","author":"Pastor-Pellicer Joan","unstructured":"Joan Pastor-Pellicer, S Espa\u00f1a-Boquera, Francisco Zamora-Mart\u00ednez, M\u00a0Zeshan Afzal, and Maria\u00a0Jose Castro-Bleda. 2015. Insights on the use of convolutional neural networks for document image binarization. In International Work-Conference on Artificial Neural Networks. Springer, 115\u2013126."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFHR.2010.118"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFHR.2012.216"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFHR-2018.2018.00091"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICFHR.2016.0118"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00249"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2011.299"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.219"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3286826"},{"key":"e_1_3_2_1_41_1","volume-title":"Adaptive document image binarization. Pattern recognition 33, 2","author":"Sauvola Jaakko","year":"2000","unstructured":"Jaakko Sauvola and Matti Pietik\u00e4inen. 2000. Adaptive document image binarization. Pattern recognition 33, 2 (2000), 225\u2013236."},{"key":"e_1_3_2_1_42_1","volume-title":"Docentr: An end-to-end document image enhancement transformer. arXiv preprint arXiv:2201.10252","author":"Souibgui Mohamed\u00a0Ali","year":"2022","unstructured":"Mohamed\u00a0Ali Souibgui, Sanket Biswas, Sana\u00a0Khamekhem Jemni, Yousri Kessentini, Alicia Forn\u00e9s, Josep Llad\u00f3s, and Umapada Pal. 2022. Docentr: An end-to-end document image enhancement transformer. arXiv preprint arXiv:2201.10252 (2022)."},{"key":"e_1_3_2_1_43_1","volume-title":"Text-DIAE: Degradation Invariant Autoencoders for Text Recognition and Document Enhancement. arXiv preprint arXiv:2203.04814","author":"Souibgui Mohamed\u00a0Ali","year":"2022","unstructured":"Mohamed\u00a0Ali Souibgui, Sanket Biswas, Andres Mafla, Ali\u00a0Furkan Biten, Alicia Forn\u00e9s, Yousri Kessentini, Josep Llad\u00f3s, Lluis Gomez, and Dimosthenis Karatzas. 2022. Text-DIAE: Degradation Invariant Autoencoders for Text Recognition and Document Enhancement. arXiv preprint arXiv:2203.04814 (2022)."},{"key":"e_1_3_2_1_44_1","volume-title":"DE-GAN: a conditional generative adversarial network for document enhancement","author":"Souibgui Mohamed\u00a0Ali","year":"2020","unstructured":"Mohamed\u00a0Ali Souibgui and Yousri Kessentini. 2020. DE-GAN: a conditional generative adversarial network for document enhancement. IEEE Transactions on Pattern Analysis and Machine Intelligence (2020)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/1815330.1815351"},{"key":"e_1_3_2_1_46_1","volume-title":"Robust document image binarization technique for degraded document images","author":"Su Bolan","year":"2012","unstructured":"Bolan Su, Shijian Lu, and Chew\u00a0Lim Tan. 2012. Robust document image binarization technique for degraded document images. IEEE transactions on image processing 22, 4 (2012), 1408\u20131417."},{"key":"e_1_3_2_1_47_1","volume-title":"Two-stage generative adversarial networks for document image binarization with color noise and background removal. arXiv preprint arXiv:2010.10103","author":"Suh Sungho","year":"2020","unstructured":"Sungho Suh, Jihun Kim, Paul Lukowicz, and Yong\u00a0Oh Lee. 2020. Two-stage generative adversarial networks for document image binarization with color noise and background removal. arXiv preprint arXiv:2010.10103 (2020)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68787-8_21"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.25"},{"key":"e_1_3_2_1_50_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.08.025"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3127736"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1186\/s13640-021-00556-4"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2022.12.011"},{"key":"e_1_3_2_1_55_1","volume-title":"DocDiff: Document Enhancement via Residual Diffusion Models. ArXiv abs\/2305.03892","author":"Yang Zongyuan","year":"2023","unstructured":"Zongyuan Yang, Baolin Liu, Yongping Xiong, Lan Yi, Guibin Wu, Xiaojun Tang, Ziqi Liu, Junjie Zhou, and Xing Zhang. 2023. DocDiff: Document Enhancement via Residual Diffusion Models. ArXiv abs\/2305.03892 (2023)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2019.106968"},{"key":"e_1_3_2_1_57_1","volume-title":"Object detection with deep learning: A review","author":"Zhao Zhong-Qiu","year":"2019","unstructured":"Zhong-Qiu Zhao, Peng Zheng, Shou-tao Xu, and Xindong Wu. 2019. Object detection with deep learning: A review. IEEE transactions on neural networks and learning systems 30, 11 (2019), 3212\u20133232."}],"event":{"name":"ICVGIP '23: Indian Conference on Computer Vision, Graphics and Image Processing","location":"Rupnagar India","acronym":"ICVGIP '23"},"container-title":["Proceedings of the Fourteenth Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627631.3627639","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3627631.3627639","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:51:08Z","timestamp":1755892268000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627631.3627639"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,15]]},"references-count":57,"alternative-id":["10.1145\/3627631.3627639","10.1145\/3627631"],"URL":"https:\/\/doi.org\/10.1145\/3627631.3627639","relation":{},"subject":[],"published":{"date-parts":[[2023,12,15]]},"assertion":[{"value":"2024-01-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}