{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,10]],"date-time":"2026-04-10T10:04:56Z","timestamp":1775815496290,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,26]],"date-time":"2021-10-26T00:00:00Z","timestamp":1635206400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2020M681173,2021T140124"],"award-info":[{"award-number":["2020M681173,2021T140124"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Science and Technology Innovation Action Plan","award":["19511120400"],"award-info":[{"award-number":["19511120400"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,26]]},"DOI":"10.1145\/3459637.3482491","type":"proceedings-article","created":{"date-parts":[[2021,10,30]],"date-time":"2021-10-30T18:34:11Z","timestamp":1635618851000},"page":"2211-2220","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["WebKE"],"prefix":"10.1145","author":[{"given":"Chenhao","family":"Xie","sequence":"first","affiliation":[{"name":"Fudan University &amp; Shuyan Technology Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenhao","family":"Huang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaqing","family":"Liang","sequence":"additional","affiliation":[{"name":"Fudan University &amp; Shuyan Technology Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengsong","family":"Huang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanghua","family":"Xiao","sequence":"additional","affiliation":[{"name":"Fudan University &amp; Fudan-Aishu Cognitive Intelligence Joint Research Center, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,30]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.14778\/2536206.2536209"},{"key":"e_1_3_2_1_3_1","unstructured":"Joseph Paul Cohen W. Ding and A. Bagherjeiran. 2015. Semi-Supervised Web Wrapper Repair via Recursive Tree Matching. ArXiv Vol. abs\/1505.01303 (2015).  Joseph Paul Cohen W. Ding and A. Bagherjeiran. 2015. Semi-Supervised Web Wrapper Repair via Recursive Tree Matching. ArXiv Vol. abs\/1505.01303 (2015)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.14778\/1938545.1938547"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of NAACL-HLT.","author":"Devlin J.","year":"2019","unstructured":"J. Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2019 . BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding . In Proceedings of NAACL-HLT. J. Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of NAACL-HLT."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623623"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/2145432.2145596"},{"key":"e_1_3_2_1_8_1","volume-title":"LAMBERT: Layout-Aware language Modeling using BERT for information extraction. ArXiv","author":"Garncarek Lukasz","year":"2020","unstructured":"Lukasz Garncarek , Rafal Powalski , Tomasz Stanislawek , Bartosz Topolski , Piotr Halama , and Filip Grali'nski . 2020 . LAMBERT: Layout-Aware language Modeling using BERT for information extraction. ArXiv , Vol. abs\/ 2002 .08087 (2020). Lukasz Garncarek, Rafal Powalski, Tomasz Stanislawek, Bartosz Topolski, Piotr Halama, and Filip Grali'nski. 2020. LAMBERT: Layout-Aware language Modeling using BERT for information extraction. ArXiv, Vol. abs\/2002.08087 (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aimag.v36i1.2567"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2011.5767842"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2011.5767842"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2009916.2010020"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/2002472.2002541"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of EMNLP.","author":"Katti Anoop R.","unstructured":"Anoop R. Katti , C. Reisswig , Cordula Guder , Sebastian Brarda , Steffen Bickel , Johannes H\u00f6hne , and J. Faddoul . 2018. Chargrid: Towards Understanding 2D Documents . In Proceedings of EMNLP. Anoop R. Katti, C. Reisswig, Cordula Guder, Sebastian Brarda, Steffen Bickel, Johannes H\u00f6hne, and J. Faddoul. 2018. Chargrid: Towards Understanding 2D Documents. In Proceedings of EMNLP."},{"key":"e_1_3_2_1_16_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba . 2014 . Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014). Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_17_1","volume-title":"Doorenbos","author":"Kushmerick N.","year":"1997","unstructured":"N. Kushmerick , Daniel S. Weld , and Robert B . Doorenbos . 1997 . Wrapper Induction for Information Extraction. In Proceedings of IJCAI. N. Kushmerick, Daniel S. Weld, and Robert B. Doorenbos. 1997. Wrapper Induction for Information Extraction. In Proceedings of IJCAI."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of ACL.","author":"Li Xiaoya","unstructured":"Xiaoya Li , Fan Yin , Zijun Sun , Xiayu Li , Arianna Yuan , Duo Chai , Mingxin Zhou , and J. Li . 2019. Entity-Relation Extraction as Multi-Turn Question Answering . In Proceedings of ACL. Xiaoya Li, Fan Yin, Zijun Sun, Xiayu Li, Arianna Yuan, Duo Chai, Mingxin Zhou, and J. Li. 2019. Entity-Relation Extraction as Multi-Turn Question Answering. In Proceedings of ACL."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3172077.3172413"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403153"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.14778\/3231751.3231758"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of NAACL.","author":"Lockard Colin","unstructured":"Colin Lockard , Prashant Shiralkar , and X. Dong . 2019. OpenCeres: When Open Information Extraction Meets the Semi-Structured Web . In Proceedings of NAACL. Colin Lockard, Prashant Shiralkar, and X. Dong. 2019. OpenCeres: When Open Information Extraction Meets the Semi-Structured Web. In Proceedings of NAACL."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.721"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-010-0137-1"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.5555\/1690219.1690287"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-017-1097-2"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.14778\/2824032.2824120"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2016.7498320"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11431-020-1647-3"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052708"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33017072"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1265"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of ICLR.","author":"Wang Shuohang","year":"2017","unstructured":"Shuohang Wang and Jing Jiang . 2017 . Machine comprehension using match-lstm and answer pointer . In Proceedings of ICLR. Shuohang Wang and Jing Jiang. 2017. Machine comprehension using match-lstm and answer pointer. In Proceedings of ICLR."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.5555\/3304222.3304390"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1599"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.136"},{"key":"e_1_3_2_1_38_1","unstructured":"Chenhao Xie Qiao Cheng Jiaqing Liang Lihan Chen and Y. Xiao. 2020. Collective Loss Function for Positive and Unlabeled Learning. ArXiv Vol. abs\/2005.03228 (2020).  Chenhao Xie Qiao Cheng Jiaqing Liang Lihan Chen and Y. Xiao. 2020. Collective Loss Function for Positive and Unlabeled Learning. ArXiv Vol. abs\/2005.03228 (2020)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.277"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of SIGKDD","author":"Xu Yiheng","year":"2020","unstructured":"Yiheng Xu , Minghao Li , Lei Cui , Shaohan Huang , Furu Wei , and M. Zhou . 2020. LayoutLM: Pre-training of Text and Layout for Document Image Understanding . Proceedings of SIGKDD ( 2020 ). Yiheng Xu, Minghao Li, Lei Cui, Shaohan Huang, Furu Wei, and M. Zhou. 2020. LayoutLM: Pre-training of Text and Layout for Document Image Understanding. Proceedings of SIGKDD (2020)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.5555\/1614164.1614177"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1047"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-007-0022-0"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/546"}],"event":{"name":"CIKM '21: The 30th ACM International Conference on Information and Knowledge Management","location":"Virtual Event Queensland Australia","acronym":"CIKM '21","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 30th ACM International Conference on Information &amp; Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3459637.3482491","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3459637.3482491","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:50Z","timestamp":1750193330000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3459637.3482491"}},"subtitle":["Knowledge Extraction from Semi-structured Web with Pre-trained Markup Language Model"],"short-title":[],"issued":{"date-parts":[[2021,10,26]]},"references-count":43,"alternative-id":["10.1145\/3459637.3482491","10.1145\/3459637"],"URL":"https:\/\/doi.org\/10.1145\/3459637.3482491","relation":{},"subject":[],"published":{"date-parts":[[2021,10,26]]},"assertion":[{"value":"2021-10-30","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}