{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T00:50:32Z","timestamp":1725583832880},"publisher-location":"Berlin, Heidelberg","reference-count":31,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642208409"},{"type":"electronic","value":"9783642208416"}],"license":[{"start":{"date-parts":[[2011,1,1]],"date-time":"2011-01-01T00:00:00Z","timestamp":1293840000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011]]},"DOI":"10.1007\/978-3-642-20841-6_41","type":"book-chapter","created":{"date-parts":[[2011,5,27]],"date-time":"2011-05-27T12:12:54Z","timestamp":1306498374000},"page":"500-511","source":"Crossref","is-referenced-by-count":5,"title":["An Efficient Pre-processing Method to Identify Logical Components from PDF Documents"],"prefix":"10.1007","author":[{"given":"Ying","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Bai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liangcai","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"2","key":"41_CR1","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1023\/A:1009715923555","volume":"2","author":"C.J.C. Burges","year":"1998","unstructured":"Burges, C.J.C.: A tutorial on support vector machines for pattern recognition. Journal Data Mining and Knowledge Discovery\u00a02(2), 121\u2013167 (1998)","journal-title":"Journal Data Mining and Knowledge Discovery"},{"key":"41_CR2","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-540-28640-0_20","volume-title":"Document Analysis Systems VI","author":"H. Chao","year":"2004","unstructured":"Chao, H., Fan, J.: Layout and content extraction for PDF documents. In: Marinai, S., Dengel, A.R. (eds.) DAS 2004. LNCS, vol.\u00a03163, pp. 213\u2013224. Springer, Heidelberg (2004)"},{"key":"41_CR3","doi-asserted-by":"crossref","unstructured":"Chen, S.T.H., Tsai, J.: Mining tables from large scale html texts. In: Proc. 18th International Conference Computational Liguistics, Saarbrucken, Germany (2000)","DOI":"10.3115\/990820.990845"},{"key":"41_CR4","unstructured":"Ha, J., Haralick, R., Philips, I.: Recursive x-y cut using bounding boxes of connected components. In: Proc. Third International Conference Document Analysis and Recognition, pp. 952\u2013955 (1955)"},{"key":"41_CR5","unstructured":"Hurst, M.: Layout and language: Challenges for table understanding on the web. In: Proceedings of the International Workshop on Web Document Analysis, pp. 27\u201330 (2001)"},{"key":"41_CR6","unstructured":"Shin, N.G.J.: Table recognition and evaluation. In: Proceeding of the Class of 2005 Senior Conference, Computer Science Department, Swarthmore College, pp. 8\u201313 (2005)"},{"key":"41_CR7","unstructured":"Joachims, T.: Svm light, http:\/\/svmlight.joachims.org\/"},{"key":"41_CR8","doi-asserted-by":"crossref","unstructured":"Kieninger, T., Dengel, A.: Applying the t-rec table recognition system to the business letter domain. In: In Proc. of the 6th International Conference on Document Analysis and Recognition, pp. 518\u2013522 (September 2001)","DOI":"10.1109\/ICDAR.2001.953843"},{"key":"41_CR9","doi-asserted-by":"crossref","unstructured":"Kieninger, T.G.: Table structure recognition based on robust block segmentation. In: Proceeding of Document Recognition V, SPIE, vol.\u00a03305, pp. 22\u201332 (January 1998)","DOI":"10.1117\/12.304642"},{"key":"41_CR10","doi-asserted-by":"crossref","unstructured":"Krupl, B., Herzog, M., Gatterbauer, W.: Using visual cues for extraction of tabular data from arbitrary html documents. In: Proceeding of the 14th International Conference on World Wide Web, pp. 1000\u20131001 (2005)","DOI":"10.1145\/1062745.1062838"},{"key":"41_CR11","first-page":"282","volume-title":"Proc. 18th ICML","author":"J. Lafferty","year":"2001","unstructured":"Lafferty, J., McCallum, A., Pereira, F.: Conditional random fields: Probabilistic models for segmenting and labeling sequence data. In: Proc. 18th ICML, pp. 282\u2013289. Morgan Kaufmann, San Francisco (2001)"},{"key":"41_CR12","doi-asserted-by":"crossref","unstructured":"Liu, Y., Bai, K., Mitra, P., Giles, C.L.: TableSeer: Automatic Table Metadata Extraction and Searching in Digital Libraries. In: ACM\/IEEE Joint Conference on Digital Libraries, JCDL, pp. 91\u2013100 (2007)","DOI":"10.1145\/1255175.1255193"},{"key":"41_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Y., Bai, K., Mitra, P., Giles, C.L.: Improving the table boundary detection in pdfs by fixing the sequence error of the sparse lines. In: 10th International Conference on Document Analysis and Recognition, ICDAR 2009 (2009)","DOI":"10.1109\/ICDAR.2009.138"},{"key":"41_CR14","doi-asserted-by":"crossref","unstructured":"Liu, Y., Mitra, P., Giles, C.L.: Identifying Table Boundaries in Digital Documents via Sparse Line Detection. In: CIKM 2008, Napa Valley, California (2008)","DOI":"10.1145\/1458082.1458255"},{"key":"41_CR15","unstructured":"McCallum, A.: Efficiently inducing features of conditional random fields. In: Nineteenth Conference on UAI (2003)"},{"key":"41_CR16","doi-asserted-by":"crossref","unstructured":"McCallum, A., Li, W.: Early results for named entity recognition with conditional random fields. In: CONLL 2003 Proceedings of the Seventh Conference on Natural Language Learning at HLT-NAACL 2003, vol.\u00a04 (2003)","DOI":"10.3115\/1119176.1119206"},{"key":"41_CR17","doi-asserted-by":"crossref","unstructured":"Ng, H., Lim, C., Koo, J.: Learning to recognize tables in free text. In: ACL Proceedings of the 37th Annual Meeting of the Association for Computational Linguistics on Computational Linguistics (1999)","DOI":"10.3115\/1034678.1034746"},{"key":"41_CR18","doi-asserted-by":"crossref","unstructured":"Ng, H., Lim, C.Y., Koo, J.T.: Learning to recognize tables in free text. In: Proc. of the 37th Annual Meeting of the Association of Computational Linguistics on Computational Linguistics, pp. 443\u2013450 (1999)","DOI":"10.3115\/1034678.1034746"},{"key":"41_CR19","doi-asserted-by":"crossref","unstructured":"Penn, G., Hu, J., Luo, H., McDonald, R.: Flexible web document analysis for delivery to narrow-bandwidth devices. In: Sixth International Conference on Document Analysis and Recognition (2001)","DOI":"10.1109\/ICDAR.2001.953951"},{"key":"41_CR20","doi-asserted-by":"crossref","unstructured":"Pinto, D., McCallum, A., Wei, X., Bruce, W.: Table extraction using conditional random fields. In: Proceeding of Proceedings of the 26th ACM SIGIR, Toronto, Canada (July 2003)","DOI":"10.1145\/860435.860479"},{"issue":"3","key":"41_CR21","first-page":"660","volume":"21","author":"S. Safavian","year":"1991","unstructured":"Safavian, S., Landgrebe, D.: A survey of decision tree classifier methodology. SMC\u00a021(3), 660\u2013674 (1991)","journal-title":"SMC"},{"key":"41_CR22","doi-asserted-by":"crossref","unstructured":"Sha, F., Pereira, F.: Shallow parsing with conditional random fields. In: NAACL 2003, Proceedings of the 2003 Conference of the North American Chapter of the Association for Computational Linguistics on Human Language Technology, vol. 1 (2003)","DOI":"10.3115\/1073445.1073473"},{"key":"41_CR23","doi-asserted-by":"crossref","unstructured":"Shamilian, J., Baird, H., Wood, T.: A retargetable table reader. In: Proc. of the 4th Int\u2019l Conf. on Document Analysis and Recognition, pp. 158\u2013163 (1997)","DOI":"10.1109\/ICDAR.1997.619833"},{"key":"41_CR24","doi-asserted-by":"crossref","unstructured":"Stoffel, A., Spretke, D., Kinnemann, H., Keim, D.A.: Enhancing document structure analysis using visual analytics. In: SAC 2010 Proceedings of the 2010 ACM Symposium on Applied Computing (2010)","DOI":"10.1145\/1774088.1774091"},{"key":"41_CR25","doi-asserted-by":"crossref","unstructured":"Wang, J., Hu, J.: A machine learning based approach for table detection on the web. In: The Eleventh International World Wide Web Conference 2002, pp. 242\u2013250 (November 2002)","DOI":"10.1145\/511446.511478"},{"key":"41_CR26","doi-asserted-by":"crossref","unstructured":"Wang, Y., Hu, J.: Detecting tables in html documents. In: Proc. of the 5th IAPR DAS, Princeton, NJ (2002)","DOI":"10.1007\/3-540-45869-7_29"},{"key":"41_CR27","unstructured":"Wang, Y., Philips, I., Haralick, R.: Automatic table ground truth generation and a background-analysis-based table structure extraction method. In: Proc. of the 6th Int\u2019l Conference on Document Analysis and Recognition, p. 528 (September 2001)"},{"key":"41_CR28","unstructured":"Yildiz, B., Kaiser, K., Miksch, S.: pdf2table: A method to extract table information from pdf files. In: Proceedings of the 2nd Indian International Conference on Artificial Intelligence IICAI 2005, Pune, India (2005)"},{"key":"41_CR29","unstructured":"Yoshida, M., Torisawa, K., Tsujii, J.: A method to integrate tables of the world wide web. In: Proceedings of the International Workshop on Web Document Analysis (WDA 2001) (2001)"},{"issue":"1","key":"41_CR30","first-page":"1","volume":"7","author":"R. Zanibbi","year":"2004","unstructured":"Zanibbi, R., Blostein, D., Cordy, J.: A survey of table recognition: Models, observations, transformations, and inferences. Int\u2019l J. Document Analysis and Recognition\u00a07(1), 1\u201316 (2004)","journal-title":"Int\u2019l J. Document Analysis and Recognition"},{"key":"41_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"196","DOI":"10.1007\/BFb0026690","volume-title":"Machine Learning: ECML-98","author":"Z. Zheng","year":"1998","unstructured":"Zheng, Z.: Naive bayesian classifier committees. In: N\u00e9dellec, C., Rouveirol, C. (eds.) ECML 1998. LNCS, vol.\u00a01398, pp. 196\u2013207. Springer, Heidelberg (1998)"}],"container-title":["Lecture Notes in Computer Science","Advances in Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-20841-6_41","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,6,11]],"date-time":"2019-06-11T07:29:04Z","timestamp":1560238144000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-20841-6_41"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011]]},"ISBN":["9783642208409","9783642208416"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-20841-6_41","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2011]]}}}