{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:58:08Z","timestamp":1785488288782,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":67,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774550","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["From Words to Paragraphs: Hierarchical Dense Text Detection with SAM-Adaptive Backbone"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-1683-3936","authenticated-orcid":false,"given":"Shashank Krishna","family":"Vempati","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, IIT Delhi, New Delhi, Delhi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4430-186X","authenticated-orcid":false,"given":"Gaurav","family":"Talebailkar","sequence":"additional","affiliation":[{"name":"IIT Delhi, New Delhi, Delhi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6737-5585","authenticated-orcid":false,"given":"Sai Prabhath","family":"Bogam","sequence":"additional","affiliation":[{"name":"IIT Delhi, New Delhi, Delhi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4456-7175","authenticated-orcid":false,"given":"Bhaskar","family":"Arun","sequence":"additional","affiliation":[{"name":"Tata 1mg, Gurugram, Haryana, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1882-8363","authenticated-orcid":false,"given":"Kayalvizhi","family":"Ganesan","sequence":"additional","affiliation":[{"name":"Tata 1mg, Gurugram, Haryana, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0155-0250","authenticated-orcid":false,"given":"Chetan","family":"Arora","sequence":"additional","affiliation":[{"name":"IIT Delhi, New Delhi, Delhi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"2024. ICDAR 2024 Occluded RoadText Competition. https:\/\/rrc.cvc.uab.es\/?ch=29&com=introduction"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00959"},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i2.27844"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.157"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"crossref","unstructured":"Chee-Kheng Chng Yuliang Liu Yipeng Sun Chun\u00a0Chet Ng Canjie Luo Zihan Ni ChuanMing Fang Shuaitao Zhang Junyu Han Errui Ding Jingtuo Liu Dimosthenis Karatzas Chee\u00a0Seng Chan and Lianwen Jin. 2019. ICDAR2019 Robust Reading Challenge on Arbitrary-Shaped Text (RRC-ArT). arxiv:https:\/\/arXiv.org\/abs\/1909.07145\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1909.07145","DOI":"10.1109\/ICDAR.2019.00252"},{"key":"e_1_3_3_3_8_2","unstructured":"Alexey Dosovitskiy. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.11929 (2020)."},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"crossref","unstructured":"Bo Du Jian Ye Jing Zhang Juhua Liu and Dacheng Tao. 2022. I3cl: Intra-and inter-instance collaborative learning for arbitrary-shaped scene text detection. International Journal of Computer Vision 130 8 (2022) 1961\u20131977.","DOI":"10.1007\/s11263-022-01616-6"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"crossref","unstructured":"Pedro\u00a0F Felzenszwalb Ross\u00a0B Girshick David McAllester and Deva Ramanan. 2009. Object detection with discriminatively trained part-based models. IEEE transactions on pattern analysis and machine intelligence 32 9 (2009) 1627\u20131645.","DOI":"10.1109\/TPAMI.2009.167"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.254"},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"crossref","unstructured":"Minghang He Minghui Liao Zhibo Yang Humen Zhong Jun Tang Wenqing Cheng Cong Yao Yongpan Wang and Xiang Bai. 2021. MOST: A Multi-Oriented Scene Text Detector with Localization Refinement. 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021) 8809\u20138818. https:\/\/api.semanticscholar.org\/CorpusID:233004432","DOI":"10.1109\/CVPR46437.2021.00870"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"crossref","unstructured":"Pan He Weilin Huang Tong He Qile Zhu Yu Qiao and Xiaolin Li. 2017. Single Shot Text Detector with Regional Attention. 2017 IEEE International Conference on Computer Vision (ICCV) (2017) 3066\u20133074. https:\/\/api.semanticscholar.org\/CorpusID:23968407","DOI":"10.1109\/ICCV.2017.331"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"crossref","unstructured":"Wenhao He Xu-Yao Zhang Fei Yin and Cheng-Lin Liu. 2017. Deep Direct Regression for Multi-oriented Scene Text Detection. 2017 IEEE International Conference on Computer Vision (ICCV) (2017) 745\u2013753. https:\/\/api.semanticscholar.org\/CorpusID:1339502","DOI":"10.1109\/ICCV.2017.87"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Lei Ke Mingqiao Ye Martin Danelljan Yu-Wing Tai Chi-Keung Tang Fisher Yu et\u00a0al. 2024. Segment anything in high quality. Advances in Neural Information Processing Systems 36 (2024).","DOI":"10.52202\/075280-1303"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/VLSID.2019.00058"},{"key":"e_1_3_3_3_18_2","unstructured":"Diederik\u00a0P Kingma. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1412.6980 (2014)."},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"crossref","unstructured":"Alexander Kirillov Kaiming He Ross\u00a0B. Girshick Carsten Rother and Piotr Doll\u00e1r. 2018. Panoptic Segmentation. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2018) 9396\u20139405. https:\/\/api.semanticscholar.org\/CorpusID:4853375","DOI":"10.1109\/CVPR.2019.00963"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"crossref","unstructured":"Harold\u00a0W Kuhn. 1955. The Hungarian method for the assignment problem. Naval research logistics quarterly 2 1-2 (1955) 83\u201397.","DOI":"10.1002\/nav.3800020109"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01483"},{"key":"e_1_3_3_3_23_2","volume-title":"AAAI Conference on Artificial Intelligence","author":"Liao Minghui","year":"2016","unstructured":"Minghui Liao, Baoguang Shi, Xiang Bai, Xinggang Wang, and Wenyu Liu. 2016. TextBoxes: A Fast Text Detector with a Single Deep Neural Network. In AAAI Conference on Artificial Intelligence. https:\/\/api.semanticscholar.org\/CorpusID:16796292"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"crossref","unstructured":"Minghui Liao Zhisheng Zou Zhaoyi Wan Cong Yao and Xiang Bai. 2022. Real-Time Scene Text Detection With Differentiable Binarization and Adaptive Scale Fusion. IEEE Transactions on Pattern Analysis and Machine Intelligence 45 (2022) 919\u2013931. https:\/\/api.semanticscholar.org\/CorpusID:260432782","DOI":"10.1109\/TPAMI.2022.3155612"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00983"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00112"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_2"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_3_3_30_2","unstructured":"Shaoqing Ren. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1506.01497 (2015)."},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"crossref","unstructured":"Baoguang Shi Xiang Bai and Serge\u00a0J. Belongie. 2017. Detecting Oriented Text in Natural Images by Linking Segments. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017) 3482\u20133490. https:\/\/api.semanticscholar.org\/CorpusID:545733","DOI":"10.1109\/CVPR.2017.371"},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00869"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00250"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"crossref","unstructured":"Jun Tang Zhibo Yang Yongpan Wang Qi Zheng Yongchao Xu and Xiang Bai. 2019. SegLink++: Detecting Dense and Arbitrary-shaped Scene Text by Instance-aware Component Grouping. Pattern Recognit. 96 (2019). https:\/\/api.semanticscholar.org\/CorpusID:198465665","DOI":"10.1016\/j.patcog.2019.06.020"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00452"},{"key":"e_1_3_3_3_36_2","unstructured":"Zhi Tian Weilin Huang Tong He Pan He and Yu Qiao. 2016. Detecting Text in Natural Image with Connectionist Text Proposal Network. ArXiv abs\/1609.03605 (2016). https:\/\/api.semanticscholar.org\/CorpusID:14728290"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"crossref","unstructured":"Zhuotao Tian Michelle Shu Pengyuan Lyu Ruiyu Li Chao Zhou Xiaoyong Shen and Jiaya Jia. 2019. Learning Shape-Aware Embedding for Scene Text Detection. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019) 4229\u20134238. https:\/\/api.semanticscholar.org\/CorpusID:198162230","DOI":"10.1109\/CVPR.2019.00436"},{"key":"e_1_3_3_3_38_2","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_3_3_39_2","unstructured":"Andreas Veit Tomas Matera Luk\u00e1s Neumann Jiri Matas and Serge\u00a0J. Belongie. 2016. COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images. ArXiv abs\/1601.07140 (2016). https:\/\/api.semanticscholar.org\/CorpusID:2838551"},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"crossref","unstructured":"Runmin Wang Yanbin Zhu Hua Chen Zhenlin Zhu Xiangyu Zhang Yajun Ding Shengyou Qian Changxin Gao Li Liu and Nong Sang. 2024. TTDNet: An End-to-End Traffic Text Detection Framework for Open Driving Environments. IEEE Transactions on Intelligent Transportation Systems (2024). https:\/\/api.semanticscholar.org\/CorpusID:273594359","DOI":"10.1109\/TITS.2024.3479884"},{"key":"e_1_3_3_3_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00956"},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00853"},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"crossref","unstructured":"Xiaobing Wang Yingying Jiang Zhenbo Luo Cheng-Lin Liu Hyunsoo Choi and Sungjin Kim. 2019. Arbitrary Shape Scene Text Detection With Adaptive Text Region Representation. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019) 6442\u20136451. https:\/\/api.semanticscholar.org\/CorpusID:155093259","DOI":"10.1109\/CVPR.2019.00661"},{"key":"e_1_3_3_3_44_2","unstructured":"Xinlong Wang Rufeng Zhang Tao Kong Lei Li and Chunhua Shen. 2020. Solov2: Dynamic and fast instance segmentation. Advances in Neural information processing systems 33 (2020) 17721\u201317732."},{"key":"e_1_3_3_3_45_2","unstructured":"Junde Wu Wei Ji Yuanpei Liu Huazhu Fu Min Xu Yanwu Xu and Yueming Jin. 2023. Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.12620 (2023)."},{"key":"e_1_3_3_3_46_2","unstructured":"Enze Xie Yuhang Zang Shuai Shao Gang Yu Cong Yao and Guangyao Li. 2018. Scene Text Detection with Supervised Pyramid Context Network. ArXiv abs\/1811.08605 (2018). https:\/\/api.semanticscholar.org\/CorpusID:53790237"},{"key":"e_1_3_3_3_47_2","unstructured":"Zhaozhi Xie Bochen Guan Weihao Jiang Muyang Yi Yue Ding Hongtao Lu and Lei Zhang. 2024. PA-SAM: Prompt Adapter SAM for High-Quality Image Segmentation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.13051 (2024)."},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"crossref","unstructured":"Yongchao Xu Yukang Wang Wei Zhou Yongpan Wang Zhibo Yang and Xiang Bai. 2018. TextField: Learning a Deep Direction Field for Irregular Scene Text Detection. IEEE Transactions on Image Processing 28 (2018) 5566\u20135579. https:\/\/api.semanticscholar.org\/CorpusID:54446029","DOI":"10.1109\/TIP.2019.2900589"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/139"},{"key":"e_1_3_3_3_50_2","unstructured":"Yunhan Yang Xiaoyang Wu Tong He Hengshuang Zhao and Xihui Liu. 2023. Sam3d: Segment anything in 3d scenes. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.03908 (2023)."},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.5555\/2354409.2354851"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/72"},{"key":"e_1_3_3_3_53_2","unstructured":"Maoyuan Ye Jing Zhang Juhua Liu Chenyu Liu Baocai Yin Cong Liu Bo Du and Dacheng Tao. 2024. Hi-SAM: Marrying Segment Anything Model for Hierarchical Text Segmentation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.17904 (2024)."},{"key":"e_1_3_3_3_54_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25430"},{"key":"e_1_3_3_3_55_2","unstructured":"Liu Yuliang Jin Lianwen Zhang Shuaitao and Zhang Sheng. 2017. Detecting curve text in the wild: New dataset and new solution. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1712.02170 (2017)."},{"key":"e_1_3_3_3_56_2","unstructured":"Yu-Xiang Zeng Jun-Wei Hsieh Xin Li and Ming-Ching Chang. 2023. MixNet: Toward Accurate Detection of Challenging Scene Text in the Wild. arxiv:https:\/\/arXiv.org\/abs\/2308.12817\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2308.12817"},{"key":"e_1_3_3_3_57_2","unstructured":"Chaoning Zhang Dongshen Han Yu Qiao Jung\u00a0Uk Kim Sung-Ho Bae Seungkyu Lee and Choong\u00a0Seon Hong. 2023. Faster segment anything: Towards lightweight sam for mobile applications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.14289 (2023)."},{"key":"e_1_3_3_3_58_2","doi-asserted-by":"crossref","unstructured":"Chengquan Zhang Borong Liang Zuming Huang Mengyi En Junyu Han Errui Ding and Xinghao Ding. 2019. Look More Than Once: An Accurate Detector for Text of Arbitrary Shapes. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019) 10544\u201310553. https:\/\/api.semanticscholar.org\/CorpusID:119297349","DOI":"10.1109\/CVPR.2019.01080"},{"key":"e_1_3_3_3_59_2","doi-asserted-by":"crossref","unstructured":"Chongsheng Zhang Yuefeng Tao Kai Du Weiping Ding Wang Bin Ji Liu and Wei Wang. 2022. Character-Level Street View Text Spotting Based on Deep Multisegmentation Network for Smarter Autonomous Driving. IEEE Transactions on Artificial Intelligence 3 (2022) 297\u2013308. https:\/\/api.semanticscholar.org\/CorpusID:244332032","DOI":"10.1109\/TAI.2021.3116216"},{"key":"e_1_3_3_3_60_2","unstructured":"Kaidong Zhang and Dong Liu. 2023. Customized segment anything model for medical image segmentation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.13785 (2023)."},{"key":"e_1_3_3_3_61_2","doi-asserted-by":"publisher","unstructured":"Shi-Xue Zhang Chun Yang Xiaobin Zhu and Xu-Cheng Yin. 2024. Arbitrary Shape Text Detection via Boundary Transformer. IEEE Transactions on Multimedia 26 (2024) 1747\u20131760. 10.1109\/TMM.2023.3286657","DOI":"10.1109\/TMM.2023.3286657"},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"crossref","unstructured":"Shi-Xue Zhang Xiaobin Zhu Jie-Bo Hou Chang Liu Chun Yang Hongfa Wang and Xu-Cheng Yin. 2020. Deep Relational Reasoning Graph Network for Arbitrary Shape Text Detection. 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020) 9696\u20139705. https:\/\/api.semanticscholar.org\/CorpusID:212737226","DOI":"10.1109\/CVPR42600.2020.00972"},{"key":"e_1_3_3_3_63_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00134"},{"key":"e_1_3_3_3_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00930"},{"key":"e_1_3_3_3_65_2","doi-asserted-by":"crossref","unstructured":"Zheng Zhang Chengquan Zhang Wei Shen Cong Yao Wenyu Liu and Xiang Bai. 2016. Multi-oriented Text Detection with Fully Convolutional Networks. 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016) 4159\u20134167. https:\/\/api.semanticscholar.org\/CorpusID:2214682","DOI":"10.1109\/CVPR.2016.451"},{"key":"e_1_3_3_3_66_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.283"},{"key":"e_1_3_3_3_67_2","unstructured":"Xizhou Zhu Weijie Su Lewei Lu Bin Li Xiaogang Wang and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.04159 (2020)."},{"key":"e_1_3_3_3_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00314"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774550","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:08:40Z","timestamp":1785485320000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774550"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":67,"alternative-id":["10.1145\/3774521.3774550","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774550","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}