{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:56:51Z","timestamp":1774360611182,"version":"3.50.1"},"publisher-location":"Cham","reference-count":24,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212993","type":"print"},{"value":"9783032213006","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21300-6_14","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:03:31Z","timestamp":1774357411000},"page":"222-236","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Website Segmentation Beyond Structure: A Benchmark on\u00a0Functional and\u00a0Digital Maturity Classes"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-1263-5757","authenticated-orcid":false,"given":"Jasmin","family":"Saxer","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6751-7944","authenticated-orcid":false,"given":"Jonathan","family":"Gerber","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8385-2537","authenticated-orcid":false,"given":"Andreas","family":"Weiler","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1609-2221","authenticated-orcid":false,"given":"Michael","family":"Grossniklaus","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"14_CR1","series-title":"Communications in Computer and Information Science","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1007\/978-981-15-6168-9_17","volume-title":"Computational Linguistics","author":"JJ Andrew","year":"2020","unstructured":"Andrew, J.J., Ferrari, S., Maurel, F., Dias, G., Giguet, E.: Model-driven web page segmentation for non visual access. In: Nguyen, L.-M., Phan, X.-H., Hasida, K., Tojo, S. (eds.) PACLING 2019. CCIS, vol. 1215, pp. 191\u2013205. Springer, Singapore (2020). https:\/\/doi.org\/10.1007\/978-981-15-6168-9_17"},{"key":"14_CR2","unstructured":"Cai, D., Yu, S., Wen, J.R., Ma, W.Y.: VIPS: a vision-based page segmentation algorithm (2003)"},{"key":"14_CR3","doi-asserted-by":"publisher","unstructured":"Chen, K., et al.: MMDetection: open MMLab detection toolbox and benchmark. arXiv arXiv:1906.07155 [cs, eess], June 2019. https:\/\/doi.org\/10.48550\/arXiv.1906.07155","DOI":"10.48550\/arXiv.1906.07155"},{"key":"14_CR4","doi-asserted-by":"publisher","unstructured":"Dai, L., Ke, Z., Silamu, W.: YOLO-WS: a novel method for webpage segmentation. In: Proceedings of the 2023 4th International Conference on Computing, Networks and Internet of Things, CNIOT \u201923, July 2023, pp. 451\u2013456. Association for Computing Machinery, New York, NY, USA (2023). https:\/\/doi.org\/10.1145\/3603781.3603862, https:\/\/dl.acm.org\/doi\/10.1145\/3603781.3603862","DOI":"10.1145\/3603781.3603862"},{"key":"14_CR5","doi-asserted-by":"publisher","unstructured":"Feng, Z., et al.: CodeBERT: a pre-trained model for programming and natural languages. arXiv arXiv:2002.08155 [cs], September 2020. https:\/\/doi.org\/10.48550\/arXiv.2002.08155","DOI":"10.48550\/arXiv.2002.08155"},{"key":"14_CR6","doi-asserted-by":"publisher","unstructured":"Gerber, J., Saxer, J., Kreiner, B., Weiler, A.: Benchmarking state of the art website embedding methods for effective processing and analysis in the public sector, January 2025. iSSN 2693-5015. https:\/\/doi.org\/10.21203\/rs.3.rs-5664280\/v1, https:\/\/www.researchsquare.com\/article\/rs-5664280\/v1","DOI":"10.21203\/rs.3.rs-5664280\/v1"},{"key":"14_CR7","doi-asserted-by":"publisher","unstructured":"Gerber, J., Saxer, J., Rabishokr, K., Kreiner, B., Weiler, A.: WebClasSeg-25: a dual-classified webpage segmentation dataset - integrating functional and maturity-based analysis. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR \u201925, Padua, Italy, pp. 3792\u20133801. Association for Computing Machinery, New York, NY, USA (2025). https:\/\/doi.org\/10.1145\/3726302.3730309","DOI":"10.1145\/3726302.3730309"},{"key":"14_CR8","doi-asserted-by":"publisher","unstructured":"Ghaemmaghami, S.S.S., Miller, J.: Integrated-block: a new combination model to improve web page segmentation. J. Web Eng. 21(4), 1103\u20131144 (2022). https:\/\/doi.org\/10.13052\/jwe1540-9589.2146, https:\/\/ieeexplore.ieee.org\/abstract\/document\/10246922","DOI":"10.13052\/jwe1540-9589.2146"},{"key":"14_CR9","doi-asserted-by":"publisher","unstructured":"Griazev, K., Ramanauskait\u0117, S.: Web page content block identification with extended block properties. Appl. Sci. 13(9), 5680 (2023). https:\/\/doi.org\/10.3390\/app13095680, https:\/\/www.mdpi.com\/2076-3417\/13\/9\/5680","DOI":"10.3390\/app13095680"},{"key":"14_CR10","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Las Vegas, NV, USA, June 2016, pp. 770\u2013778. IEEE (2026). https:\/\/doi.org\/10.1109\/CVPR.2016.90, http:\/\/ieeexplore.ieee.org\/document\/7780459\/","DOI":"10.1109\/CVPR.2016.90"},{"key":"14_CR11","doi-asserted-by":"publisher","unstructured":"Huynh, M.H., Le, Q.T., Nguyen, V., Nguyen, T.: Web page segmentation: a DOM-structural cohesion analysis approach. In: Zhang, F., Wang, H., Barhamgi, M., Chen, L., Zhou, R. (eds.) Web Information Systems Engineering, WISE 2023, pp. 319\u2013333. Springer, Singapore (2023). https:\/\/doi.org\/10.1007\/978-981-99-7254-8_25","DOI":"10.1007\/978-981-99-7254-8_25"},{"key":"14_CR12","doi-asserted-by":"publisher","unstructured":"Jayashree, S.R., et al.: Multimodal web page segmentation using self-organized multi-objective clustering. ACM Trans. Inf. Syst. 40(3), 59:1\u201359:49 (2022). https:\/\/doi.org\/10.1145\/3480966, https:\/\/dl.acm.org\/doi\/10.1145\/3480966","DOI":"10.1145\/3480966"},{"key":"14_CR13","unstructured":"Jocher, G., Qiu, J., Chaurasia, A.: Ultralytics YOLO, January 2023. https:\/\/ultralytics.com, repository: https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"14_CR14","doi-asserted-by":"publisher","unstructured":"Kiesel, J., Kneist, F., Meyer, L., Komlossy, K., Stein, B., Potthast, M.: Web page segmentation revisited: evaluation framework and dataset. In: Proceedings of the 29th ACM International Conference on Information & Knowledge Management, CIKM \u201920, pp. 3047\u20133054. Association for Computing Machinery, New York, NY, USA (2020). https:\/\/doi.org\/10.1145\/3340531.3412782, https:\/\/dl.acm.org\/doi\/10.1145\/3340531.3412782","DOI":"10.1145\/3340531.3412782"},{"key":"14_CR15","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"62","DOI":"10.1007\/978-3-030-72240-1_5","volume-title":"Advances in Information Retrieval","author":"J Kiesel","year":"2021","unstructured":"Kiesel, J., Meyer, L., Kneist, F., Stein, B., Potthast, M.: An empirical comparison of web page segmentation algorithms. In: Hiemstra, D., Moens, M.-F., Mothe, J., Perego, R., Potthast, M., Sebastiani, F. (eds.) ECIR 2021. LNCS, vol. 12657, pp. 62\u201374. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-72240-1_5"},{"key":"14_CR16","doi-asserted-by":"publisher","unstructured":"Kirillov, A., et al.: Segment anything. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), , Paris, France, October 2023, pp. 3992\u20134003. IEEE (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00371, https:\/\/ieeexplore.ieee.org\/document\/10378323\/","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"14_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"374","DOI":"10.1007\/978-3-319-19890-3_24","volume-title":"Engineering the Web in the Big Data Era","author":"R Kreuzer","year":"2015","unstructured":"Kreuzer, R., Hage, J., Feelders, A.: A quantitative comparison of semantic web page segmentation approaches. In: Cimiano, P., Frasincar, F., Houben, G.-J., Schwabe, D. (eds.) ICWE 2015. LNCS, vol. 9114, pp. 374\u2013391. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-19890-3_24"},{"key":"14_CR18","doi-asserted-by":"publisher","unstructured":"Liu, Y., et al.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv arXiv:1907.11692 [cs], July 2019. https:\/\/doi.org\/10.48550\/arXiv.1907.11692","DOI":"10.48550\/arXiv.1907.11692"},{"key":"14_CR19","doi-asserted-by":"publisher","unstructured":"Manabe, T., Tajima, K.: Extracting logical hierarchical structure of HTML documents based on headings. Proc. VLDB Endow. 8(12), 1606\u20131617 (2015). https:\/\/doi.org\/10.14778\/2824032.2824058, https:\/\/dl.acm.org\/doi\/10.14778D\/2824032.2824058","DOI":"10.14778\/2824032.2824058"},{"key":"14_CR20","doi-asserted-by":"publisher","unstructured":"Ravi, N., et al.: SAM 2: segment anything in images and videos. arXiv arXiv:2408.00714 [cs], October 2024. https:\/\/doi.org\/10.48550\/arXiv.2408.00714","DOI":"10.48550\/arXiv.2408.00714"},{"key":"14_CR21","doi-asserted-by":"publisher","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Las Vegas, NV, USA, June 2016, pp. 779\u2013788. IEEE (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.91, http:\/\/ieeexplore.ieee.org\/document\/7780460\/","DOI":"10.1109\/CVPR.2016.91"},{"key":"14_CR22","doi-asserted-by":"publisher","unstructured":"Ren, B., Qian, Z., Sun, Y., Gao, C., Zhang, C.: WebSAM-adapter: adapting segment anything model for web page segmentation. In: Goharian, N., et al. (eds.) Advances in Information Retrieval, pp. 439\u2013454. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-56027-9_27","DOI":"10.1007\/978-3-031-56027-9_27"},{"key":"14_CR23","doi-asserted-by":"publisher","unstructured":"Sajjadi\u00a0Ghaemmaghami, S.S., Miller, J.: A new semantic approach to improve webpage segmentation. J. Web Eng. 20(4), 963\u2013992 (2021). https:\/\/doi.org\/10.13052\/jwe1540-9589.2042, https:\/\/ieeexplore.ieee.org\/document\/10246784","DOI":"10.13052\/jwe1540-9589.2042"},{"key":"14_CR24","doi-asserted-by":"publisher","unstructured":"Yang, A., et al.: Qwen3 Technical Report (2025). https:\/\/doi.org\/10.48550\/arXiv.2505.09388","DOI":"10.48550\/arXiv.2505.09388"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21300-6_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:03:38Z","timestamp":1774357418000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21300-6_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212993","9783032213006"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21300-6_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}