{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T03:18:03Z","timestamp":1740107883243,"version":"3.37.3"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"20","license":[{"start":{"date-parts":[[2023,8,24]],"date-time":"2023-08-24T00:00:00Z","timestamp":1692835200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,24]],"date-time":"2023-08-24T00:00:00Z","timestamp":1692835200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2021YFD1300101"],"award-info":[{"award-number":["2021YFD1300101"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Soft Comput"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s00500-023-09076-x","type":"journal-article","created":{"date-parts":[[2023,8,24]],"date-time":"2023-08-24T13:02:26Z","timestamp":1692882146000},"page":"14631-14645","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["An efficient content extraction method for webpage based on tag-line-block analysis"],"prefix":"10.1007","volume":"27","author":[{"given":"Zeqiu","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianghui","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7267-5283","authenticated-orcid":false,"given":"Ruizhi","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,24]]},"reference":[{"key":"9076_CR1","unstructured":"Baroni M, Chantree F, Kilgarriff A et al (2008) Cleaneval: a competition for cleaning web pages. In: Proceedings of the 6th international conference on language resources and evaluation, pp 638\u2013643"},{"key":"9076_CR2","doi-asserted-by":"crossref","unstructured":"Cai D, Yu S, Wen J R, et al (2003) Extracting content structure for web pages based on visual representation. In: Proceedings of the 5th Asia-pacific web conference on web technologies and applications, pp 406\u2013417","DOI":"10.1007\/3-540-36901-5_42"},{"key":"9076_CR3","doi-asserted-by":"crossref","unstructured":"Cardoso E, Jabour I, Laber E, et al (2011) An efficient language-independent method to extract content from news webpages. In: Proceedings of the 11th ACM symposium on document engineering, pp 121\u2013128","DOI":"10.1145\/2034691.2034720"},{"key":"9076_CR4","unstructured":"Chen X (2011) Universal web content extraction based on row block distribution function. https:\/\/code.google.com\/p\/cx-extractor"},{"key":"9076_CR5","unstructured":"Crescenzi V, Mecca G, Merialdo P (2001) Roadrunner: towards automatic data extraction from large web sites. In: Proceedings of the 27th international conference on very large data bases, vol. 1, pp 109\u2013118"},{"key":"9076_CR6","doi-asserted-by":"publisher","first-page":"301","DOI":"10.1016\/j.knosys.2014.07.007","volume":"70","author":"E Ferrara","year":"2014","unstructured":"Ferrara E, De Meo P, Fiumara G et al (2014) Web data extraction, applications and techniques: a survey. Knowl-Based Syst 70:301\u2013323","journal-title":"Knowl-Based Syst"},{"key":"9076_CR7","doi-asserted-by":"publisher","first-page":"106660","DOI":"10.1016\/j.ocecoaman.2023.106660","volume":"240","author":"L Gan","year":"2023","unstructured":"Gan L, Ye B, Huang Z et al (2023) Knowledge graph construction based on ship collision accident reports to improve maritime traffic safety. Ocean Coast Manag 240:106660","journal-title":"Ocean Coast Manag"},{"key":"9076_CR8","doi-asserted-by":"crossref","unstructured":"Gibson D, Punera K, Tomkins A (2005) The volume and evolution of web page templates. In: Special interest tracks and posters of the 14th international conference on World Wide Web, pp 830\u2013839","DOI":"10.1145\/1062745.1062763"},{"key":"9076_CR9","doi-asserted-by":"crossref","unstructured":"Gottron T (2008) Combining content extraction heuristics: the CombinE system. In: Proceedings of the 10th international conference on information integration and web-based applications and services, pp 591\u2013595","DOI":"10.1145\/1497308.1497418"},{"key":"9076_CR10","first-page":"327","volume":"35","author":"Y Gu","year":"2014","unstructured":"Gu Y, Gao Y, Gao B et al (2014) Research on deep web information extraction based on template and domain ontology. Comput Eng Des 35:327\u2013332","journal-title":"Comput Eng Des"},{"key":"9076_CR11","doi-asserted-by":"crossref","unstructured":"Gupta S, Kaiser G, Neistadt D et al (2003) DOM-based content extraction of html documents. In: Proceedings of the 12th international conference on World Wide Web, pp 207\u2013214","DOI":"10.1145\/775152.775182"},{"key":"9076_CR12","doi-asserted-by":"crossref","unstructured":"Hammer J, McHugh J, Garcia-Molina H (1997) Semistructured data: the TSIMMIS experience. In: Proceedings of the 1th East-European symposium on advances in databases and information systems, vol. 1, pp 1\u201313","DOI":"10.14236\/ewic\/ADBIS1997.22"},{"key":"9076_CR13","unstructured":"IDC, Statista (2022) Volume of data\/information created, captured, copied, and consumed worldwide from 2010 to 2020, with forecasts from 2021 to 2025 (in zettabytes). https:\/\/www.statista.com\/statistics\/871513\/worldwide-data-created\/"},{"issue":"12","key":"9076_CR14","first-page":"1123","volume":"44","author":"PR Joe Dhanith","year":"2022","unstructured":"Joe Dhanith PR, Surendiran B (2022) An ontology learning based approach for focused web crawling using combined normalized pointwise mutual information and Resnik algorithm. Int J Comput Appl 44(12):1123\u20131129","journal-title":"Int J Comput Appl"},{"issue":"2","key":"9076_CR15","doi-asserted-by":"publisher","first-page":"41","DOI":"10.4018\/IJWP.2019070103","volume":"11","author":"T Karthikeyan","year":"2019","unstructured":"Karthikeyan T, Sekaran K, Ranjith D et al (2019) Personalized content extraction and text classification using effective web scraping techniques. Int J Web Port 11(2):41\u201352","journal-title":"Int J Web Port"},{"key":"9076_CR16","doi-asserted-by":"crossref","unstructured":"Laber ES, de Souza CP, Jabour IV et al (2009) A fast and simple method for extracting relevant content from news webpages. In: Proceedings of the 18th ACM conference on information and knowledge management, pp 1685\u20131688","DOI":"10.1145\/1645953.1646204"},{"key":"9076_CR17","first-page":"21","volume":"9","author":"D Liang","year":"2018","unstructured":"Liang D, Yang Y, Wei Z (2018) Information extraction of web pages based on support vector machine. Comput Mod 9:21\u201326","journal-title":"Comput Mod"},{"key":"9076_CR18","doi-asserted-by":"crossref","unstructured":"Liu L, Pu C, Han W (2000) XWRAP: an XML-enabled wrapper construction system for web information sources. In: Proceedings of the 16th international conference on data engineering, pp 611\u2013621","DOI":"10.1109\/ICDE.2000.839475"},{"key":"9076_CR19","unstructured":"Rahman A, Alam H, Hartono R (2001) Content extraction from html documents. In: Proceedings of the 1st international workshop on web document analysis, pp 1\u20134"},{"key":"9076_CR20","doi-asserted-by":"crossref","unstructured":"Ramakrishna M, Gowdar L, Havanur MS et al (2010) Web mining: key accomplishments, applications and future directions. In: Proceedings of the 2010 international conference on data storage and data engineering, pp 187\u2013191","DOI":"10.1109\/DSDE.2010.53"},{"key":"9076_CR21","doi-asserted-by":"crossref","unstructured":"Samuel MO, Tolulope AI, Oyejoke OO (2019) A systematic review of current trends in web content mining. In: Proceedings of the 3th international conference on science and sustainable development, vol. 1299, p 012040","DOI":"10.1088\/1742-6596\/1299\/1\/012040"},{"key":"9076_CR22","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1007\/978-981-10-3376-6_6","volume":"719","author":"KS Sandeep","year":"2018","unstructured":"Sandeep KS, Patil N (2018) A multidimensional approach to blog mining. progress in intelligent computing techniques: theory, practice, and applications. Adv Intell Syst Comput 719:51\u201358","journal-title":"Adv Intell Syst Comput"},{"issue":"7","key":"9076_CR23","doi-asserted-by":"publisher","first-page":"779","DOI":"10.1002\/int.4550080704","volume":"8","author":"S Sestito","year":"1993","unstructured":"Sestito S, Dillon T (1993) Knowledge acquisition of conjunctive rules using multilayered neural networks. Int J Intell Syst 8(7):779\u2013805","journal-title":"Int J Intell Syst"},{"key":"9076_CR24","doi-asserted-by":"crossref","unstructured":"Sun F, Song D, Liao L (2011) Dom based content extraction via text density. In: Proceedings of the 34th international ACM SIGIR conference on research and development in information retrieval, pp 245\u2013254","DOI":"10.1145\/2009916.2009952"},{"issue":"5","key":"9076_CR25","first-page":"17","volume":"18","author":"C Sun","year":"2004","unstructured":"Sun C, Guan Y (2004) A statistical approach for content extraction from web page. J Chin Inf Process 18(5):17\u201322","journal-title":"J Chin Inf Process"},{"key":"9076_CR26","doi-asserted-by":"publisher","first-page":"64085","DOI":"10.1109\/ACCESS.2018.2877592","volume":"6","author":"Z Tan","year":"2018","unstructured":"Tan Z, He C, Fang Y et al (2018) Title-based extraction of news contents for text mining. IEEE Access 6:64085\u201364095","journal-title":"IEEE Access"},{"issue":"4","key":"9076_CR27","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1177\/0894439316643050","volume":"35","author":"A Waldherr","year":"2017","unstructured":"Waldherr A, Maier D, Miltner P et al (2017) Big data, big noise: the challenge of finding issue networks on the web. Soc Sci Comput Rev 35(4):427\u2013443","journal-title":"Soc Sci Comput Rev"},{"key":"9076_CR28","doi-asserted-by":"crossref","unstructured":"Wang Q, Fang Y, Ravula A, et al (2022) Webformer: the web-page transformer for structure information extraction. In: Proceedings of the 2022 ACM web conference, pp 3124\u20133133","DOI":"10.1145\/3485447.3512032"},{"key":"9076_CR29","doi-asserted-by":"crossref","unstructured":"Weninger T, Hsu WH, Han J (2010) CETR: content extraction via tag ratios. In: Proceedings of the 19th international conference on World Wide Web, pp 971\u2013980","DOI":"10.1145\/1772690.1772789"},{"key":"9076_CR30","doi-asserted-by":"publisher","first-page":"132","DOI":"10.1016\/j.ins.2015.12.025","volume":"342","author":"Y Wu","year":"2016","unstructured":"Wu Y (2016) Language independent web news extraction system based on text detection framework. Inf Sci 342:132\u2013149","journal-title":"Inf Sci"},{"issue":"4","key":"9076_CR31","first-page":"974","volume":"25","author":"M Yu","year":"2005","unstructured":"Yu M, Chen T, Xu H (2005) Research and design of HTML parser based on page segmentation. J Comput Appl 25(4):974\u2013976","journal-title":"J Comput Appl"},{"key":"9076_CR32","volume-title":"Content extraction from webpages using machine learning","author":"H Yunis","year":"2016","unstructured":"Yunis H, Stein B, Kiesel J et al (2016) Content extraction from webpages using machine learning. Bauhaus-Universitaet Weimar"},{"key":"9076_CR33","doi-asserted-by":"publisher","first-page":"40475","DOI":"10.1109\/ACCESS.2019.2907570","volume":"7","author":"H Zhang","year":"2019","unstructured":"Zhang H, Li L, Hu W et al (2019) Visualization of location-referenced web textual information based on map mashups. IEEE Access 7:40475\u201340487","journal-title":"IEEE Access"},{"key":"9076_CR34","doi-asserted-by":"crossref","unstructured":"Zhang Z, Yu B, Liu T, et al. (2023) Learning structural co-occurrences for structured web data extraction in low-resource settings. In: Proceedings of the 2023 ACM web conference, pp 1683\u20131692","DOI":"10.1145\/3543507.3583387"}],"container-title":["Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00500-023-09076-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00500-023-09076-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00500-023-09076-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T12:17:48Z","timestamp":1693570668000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00500-023-09076-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,24]]},"references-count":34,"journal-issue":{"issue":"20","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["9076"],"URL":"https:\/\/doi.org\/10.1007\/s00500-023-09076-x","relation":{},"ISSN":["1432-7643","1433-7479"],"issn-type":[{"type":"print","value":"1432-7643"},{"type":"electronic","value":"1433-7479"}],"subject":[],"published":{"date-parts":[[2023,8,24]]},"assertion":[{"value":"29 July 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 August 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Informed consent was obtained from all individual participants included in the study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}}]}}