{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T14:18:54Z","timestamp":1787667534655,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":46,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032046161","type":"print"},{"value":"9783032046178","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T00:00:00Z","timestamp":1757894400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T00:00:00Z","timestamp":1757894400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-04617-8_4","type":"book-chapter","created":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T11:46:48Z","timestamp":1757936808000},"page":"60-77","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["LIGHT: Multi-modal Text Linking on\u00a0Historical Maps"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0815-9636","authenticated-orcid":false,"given":"Yijun","family":"Lin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1241-5583","authenticated-orcid":false,"given":"Rhett","family":"Olson","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4953-2960","authenticated-orcid":false,"given":"Junhan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8923-0130","authenticated-orcid":false,"given":"Yao-Yi","family":"Chiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2247-8174","authenticated-orcid":false,"given":"Jerod","family":"Weinman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,15]]},"reference":[{"key":"4_CR1","doi-asserted-by":"crossref","unstructured":"Appalaraju, S., Jasani, B., Kota, B.U., Xie, Y., Manmatha, R.: DocFormer: end-to-end transformer for document understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 993\u20131003 (2021)","DOI":"10.1109\/ICCV48922.2021.00103"},{"key":"4_CR2","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: BEiT: BERT pre-training of image transformers. In: International Conference on Learning Representations (2022). https:\/\/openreview.net\/forum?id=p-BhZSz59o4"},{"key":"4_CR3","doi-asserted-by":"crossref","unstructured":"Bi, T., et al.: Text grouping adapter: adapting pre-trained text detector for layout analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 28150\u201328159 (2024)","DOI":"10.1109\/CVPR52733.2024.02659"},{"key":"4_CR4","unstructured":"Cartography Associates: David Rumsey map collection. https:\/\/www.davidrumsey.com"},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Chiang, Y.Y., et\u00a0al.: GeoAI for the digitization of historical maps. In: Handbook of Geospatial Artificial Intelligence, pp. 217\u2013247. CRC Press (2023)","DOI":"10.1201\/9781003308423-11"},{"key":"4_CR6","series-title":"SpringerBriefs in Geography","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-66908-3","volume-title":"Using Historical Maps in Scientific Studies","author":"Y-Y Chiang","year":"2020","unstructured":"Chiang, Y.-Y., Duan, W., Leyk, S., Uhl, J.H., Knoblock, C.A.: Using Historical Maps in Scientific Studies. SG, Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-319-66908-3"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers), pp. 4171\u20134186 (2019)","DOI":"10.18653\/v1\/N19-1423"},{"key":"4_CR8","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L., Lu, Y., Wei, F.: LayoutLMv3: pre-training for document AI with unified text and image masking. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"4_CR9","doi-asserted-by":"crossref","unstructured":"Hwang, W., Yim, J., Park, S., Yang, S., Seo, M.: Spatial dependency parsing for semi-structured document information extraction. In: Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021, pp. 330\u2013343 (2021)","DOI":"10.18653\/v1\/2021.findings-acl.28"},{"key":"4_CR10","doi-asserted-by":"publisher","unstructured":"Kim, G., et al.: OCR-free document understanding transformer. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 498\u2013517. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_29","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"4_CR11","doi-asserted-by":"publisher","unstructured":"Kim, J., Li, Z., Lin, Y., Namgung, M., Jang, L., Chiang, Y.Y.: The mapKurator system: a complete pipeline for extracting and linking text from historical maps. In: Proceedings of the 31st ACM International Conference on Advances in Geographic Information Systems. SIGSPATIAL 2023 (2023). https:\/\/doi.org\/10.1145\/3589132.3625579","DOI":"10.1145\/3589132.3625579"},{"key":"4_CR12","unstructured":"Koroteev, M.V.: BERT: a review of applications in natural language processing and understanding. arXiv preprint arXiv:2103.11943 (2021)"},{"key":"4_CR13","doi-asserted-by":"publisher","unstructured":"Kyramargiou, E., Papakondylis, Y., Scalora, F., Dimitropoulos, D.: Changing the map in Greece and Italy: place-name changes in the nineteenth century. Hist. Rev.\/La Revue Historique 17, 205\u2013250 (2020). https:\/\/doi.org\/10.12681\/hr.27072","DOI":"10.12681\/hr.27072"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Li, X.H., Yin, F., Liu, C.L.: Page object detection from PDF document images by deep structured prediction and supervised clustering. In: 2018 24th International Conference on Pattern Recognition (ICPR), pp. 3627\u20133632. IEEE (2018)","DOI":"10.1109\/ICPR.2018.8546073"},{"key":"4_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1007\/978-3-030-57058-3_17","volume-title":"Document Analysis Systems","author":"X-H Li","year":"2020","unstructured":"Li, X.-H., Yin, F., Liu, C.-L.: Page segmentation using convolutional neural network and graphical model. In: Bai, X., Karatzas, D., Lopresti, D. (eds.) DAS 2020. LNCS, vol. 12116, pp. 231\u2013245. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-57058-3_17"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: An automatic approach for generating rich, linked geo-metadata from historical map images. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 3290\u20133298 (2020)","DOI":"10.1145\/3394486.3403381"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Li, Z., et al.: ICDAR 2024 competition on historical map text detection, recognition, and linking. In: 18th International Conference on Document Analysis and Recognition (ICDAR 2024), pp. 363\u2013380 (2024)","DOI":"10.1007\/978-3-031-70552-6_22"},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"Liang, M., Ma, J.W., Zhu, X., Qin, J., Yin, X.C.: LayoutFormer: hierarchical text detection towards scene text understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15665\u201315674 (2024)","DOI":"10.1109\/CVPR52733.2024.01483"},{"key":"4_CR19","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"4_CR20","doi-asserted-by":"crossref","unstructured":"Lin, Y., Chiang, Y.Y.: Hyper-local deformable transformers for text spotting on historical maps. In: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, pp. 5387\u20135397 (2024)","DOI":"10.1145\/3637528.3671589"},{"key":"4_CR21","doi-asserted-by":"publisher","unstructured":"Lin, Y., Li, Z., Chiang, Y.Y., Weinman, J.: Rumsey train and validation data for ICDAR 2024 MapText Competition (2024). https:\/\/doi.org\/10.5281\/zenodo.11516933","DOI":"10.5281\/zenodo.11516933"},{"key":"4_CR22","unstructured":"Lin, Y., et al.: ICDAR 2025 competition on historical map text detection, recognition, and linking. https:\/\/rrc.cvc.uab.es\/?ch=32"},{"key":"4_CR23","doi-asserted-by":"publisher","unstructured":"Lin, Y., et al.: ICDAR 2025 competition on historical map text detection, recognition, and linking. In: Document Analysis and Recognition - ICDAR 2025 (2025). https:\/\/doi.org\/10.1007\/978-3-032-04630-7_33","DOI":"10.1007\/978-3-032-04630-7_33"},{"key":"4_CR24","doi-asserted-by":"publisher","unstructured":"Lin, Y., Li, Z., Chiang, Y.Y., Weinman, J.: Rumsey test data for ICDAR 2024 MapText competition (2024). https:\/\/doi.org\/10.5281\/zenodo.10776183","DOI":"10.5281\/zenodo.10776183"},{"key":"4_CR25","unstructured":"Liu, Y., et al.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"4_CR26","doi-asserted-by":"crossref","unstructured":"Long, S., Qin, S., Fujii, Y., Bissacco, A., Raptis, M.: Hierarchical text spotter for joint text spotting and layout analysis. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 903\u2013913 (2024)","DOI":"10.1109\/WACV57701.2024.00095"},{"key":"4_CR27","doi-asserted-by":"crossref","unstructured":"Long, S., Qin, S., Panteleev, D., Bissacco, A., Fujii, Y., Raptis, M.: Towards end-to-end unified scene text detection and layout analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1049\u20131059 (2022)","DOI":"10.1109\/CVPR52688.2022.00112"},{"key":"4_CR28","doi-asserted-by":"publisher","unstructured":"Long, S., Qin, S., Panteleev, D., Bissacco, A., Fujii, Y., Raptis, M.: ICDAR 2023 competition on hierarchical text detection and recognition. In: Fink, G.A., Jain, R., Kise, K., Zanibbi, R. (eds.)International Conference on Document Analysis and Recognition, pp. 483\u2013497. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-41679-8_28","DOI":"10.1007\/978-3-031-41679-8_28"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Luo, C., Cheng, C., Zheng, Q., Yao, C.: GeoLayoutLM: geometric pre-training for visual information extraction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7092\u20137101 (2023)","DOI":"10.1109\/CVPR52729.2023.00685"},{"key":"4_CR30","unstructured":"Luo, S., Ding, Y., Long, S., Poon, J., Han, S.C.: Doc-GCN: heterogeneous graph convolutional networks for document layout analysis. In: Proceedings of the 29th International Conference on Computational Linguistics, pp. 2906\u20132916. International Committee on Computational Linguistics (2022). https:\/\/aclanthology.org\/2022.coling-1.256\/"},{"key":"4_CR31","doi-asserted-by":"crossref","unstructured":"Olson, R., Kim, J., Chiang, Y.Y.: Automatic search of multiword place names on historical maps. In: Proceedings of the 3rd ACM SIGSPATIAL International Workshop on Searching and Mining Large Collections of Geospatial Data, pp. 9\u201312 (2024)","DOI":"10.1145\/3681769.3698577"},{"key":"4_CR32","doi-asserted-by":"crossref","unstructured":"Olson, R.M., Kim, J., Chiang, Y.Y.: An automatic approach to finding geographic name changes on historical maps. In: Proceedings of the 31st ACM International Conference on Advances in Geographic Information Systems, pp.\u00a01\u20132 (2023)","DOI":"10.1145\/3589132.3628368"},{"key":"4_CR33","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: GloVe: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"4_CR34","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"732","DOI":"10.1007\/978-3-030-86331-9_47","volume-title":"Document Analysis and Recognition \u2013 ICDAR 2021","author":"R Powalski","year":"2021","unstructured":"Powalski, R., Borchmann, \u0141, Jurkiewicz, D., Dwojak, T., Pietruszka, M., Pa\u0142ka, G.: Going full-TILT boogie on document understanding with text-image-layout transformer. In: Llad\u00f3s, J., Lopresti, D., Uchida, S. (eds.) ICDAR 2021. LNCS, vol. 12822, pp. 732\u2013747. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-86331-9_47"},{"issue":"6","key":"4_CR35","doi-asserted-by":"publisher","first-page":"1389","DOI":"10.1002\/j.1538-7305.1957.tb01515.x","volume":"36","author":"RC Prim","year":"1957","unstructured":"Prim, R.C.: Shortest connection networks and some generalizations. Bell Syst. Technical J. 36(6), 1389\u20131401 (1957)","journal-title":"Bell Syst. Technical J."},{"key":"4_CR36","unstructured":"Ramesh, A., et al.: Zero-shot text-to-image generation. In: International Conference on Machine Learning, pp. 8821\u20138831 (2021)"},{"key":"4_CR37","doi-asserted-by":"publisher","unstructured":"Wang, J., Hu, K., Huo, Q.: DLAFormer: An end-to-end transformer for document layout analysis. In: Barney Smith, E.H., Liwicki, M., Peng, L. (eds.) International Conference on Document Analysis and Recognition, pp. 40\u201357. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-70546-5_3","DOI":"10.1007\/978-3-031-70546-5_3"},{"key":"4_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110836","volume":"156","author":"J Wang","year":"2024","unstructured":"Wang, J., Hu, K., Zhong, Z., Sun, L., Huo, Q.: Detect-order-construct: a tree construction based approach for hierarchical document structure analysis. Pattern Recogn. 156, 110836 (2024)","journal-title":"Pattern Recogn."},{"key":"4_CR39","doi-asserted-by":"publisher","unstructured":"Wang, J., et al.: Dynamic relation transformer for contextual text block detection. In: Barney Smith, E.H., Liwicki, M., Peng, L. (eds.) International Conference on Document Analysis and Recognition, pp. 313\u2013330. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-70533-5_19","DOI":"10.1007\/978-3-031-70533-5_19"},{"key":"4_CR40","doi-asserted-by":"crossref","unstructured":"Wang, R., Fujii, Y., Popat, A.C.: Post-OCR paragraph recognition by graph convolutional networks. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 493\u2013502 (2022)","DOI":"10.1109\/WACV51458.2022.00259"},{"key":"4_CR41","doi-asserted-by":"publisher","unstructured":"Xue, C., Huang, J., Zhang, W., Lu, S., Wang, C., Bai, S.: Contextual text block detection towards scene text understanding. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) European Conference on Computer Vision, pp. 374\u2013391. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19815-1_22","DOI":"10.1007\/978-3-031-19815-1_22"},{"key":"4_CR42","doi-asserted-by":"crossref","unstructured":"Ye, M., et al.: DeepSolo: let transformer decoder with explicit points solo for text spotting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19348\u201319357 (2023)","DOI":"10.1109\/CVPR52729.2023.01854"},{"key":"4_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, C., et\u00a0al.: Modeling layout reading order as ordering relations for visually-rich document understanding. In: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 9658\u20139678 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.540"},{"key":"4_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, X., Su, Y., Tripathi, S., Tu, Z.: Text spotting transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9519\u20139528 (2022)","DOI":"10.1109\/CVPR52688.2022.00930"},{"key":"4_CR45","doi-asserted-by":"publisher","unstructured":"Zhong, Z., et al.: A hybrid approach to document layout analysis for heterogeneous document images. In: Fink, G.A., Jain, R., Kise, K., Zanibbi, R. (eds.) International Conference on Document Analysis and Recognition, pp. 189\u2013206. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-41734-4_12","DOI":"10.1007\/978-3-031-41734-4_12"},{"key":"4_CR46","doi-asserted-by":"publisher","unstructured":"Zou, M., Dai, T., Petitpierre, R., Vaienti, B., Kaplan, F., di\u00a0Lenardo, I.: Recognizing and sequencing multi-word texts in maps using an attentive pointer (2025). https:\/\/doi.org\/10.21203\/rs.3.rs-6330456\/v1, under review","DOI":"10.21203\/rs.3.rs-6330456\/v1"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition \u2013 ICDAR 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-04617-8_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T13:23:15Z","timestamp":1787664195000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-04617-8_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,15]]},"ISBN":["9783032046161","9783032046178"],"references-count":46,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-04617-8_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,15]]},"assertion":[{"value":"15 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Wuhan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iapr.org\/icdar2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}