{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T19:41:09Z","timestamp":1742931669721,"version":"3.40.3"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783031162091"},{"type":"electronic","value":"9783031162107"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-16210-7_11","type":"book-chapter","created":{"date-parts":[[2022,9,20]],"date-time":"2022-09-20T23:03:09Z","timestamp":1663714989000},"page":"137-149","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Multi-Wiki90k: Multilingual Benchmark Dataset for\u00a0Paragraph Segmentation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8029-6582","authenticated-orcid":false,"given":"Micha\u0142","family":"Sw\u0119drowski","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5201-0364","authenticated-orcid":false,"given":"Piotr","family":"Mi\u0142kowski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4879-8854","authenticated-orcid":false,"given":"Bart\u0142omiej","family":"Bojanowski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7665-6896","authenticated-orcid":false,"given":"Jan","family":"Koco\u0144","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,21]]},"reference":[{"key":"11_CR1","doi-asserted-by":"publisher","first-page":"597","DOI":"10.1162\/tacl_a_00288","volume":"7","author":"M Artetxe","year":"2019","unstructured":"Artetxe, M., Schwenk, H.: Massively multilingual sentence embeddings for zero-shot cross-lingual transfer and beyond. Trans. Assoc. Comput. Linguist. 7, 597\u2013610 (2019)","journal-title":"Trans. Assoc. Comput. Linguist."},{"issue":"1","key":"11_CR2","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1023\/A:1007506220214","volume":"34","author":"D Beeferman","year":"1999","unstructured":"Beeferman, D., Berger, A., Lafferty, J.: Statistical models for text segmentation. Mach. Learn. 34(1), 177\u2013210 (1999)","journal-title":"Mach. Learn."},{"issue":"9","key":"11_CR3","doi-asserted-by":"publisher","first-page":"575","DOI":"10.1145\/362342.362367","volume":"16","author":"C Bron","year":"1973","unstructured":"Bron, C., Kerbosch, J.: Algorithm 457: finding all cliques of an undirected graph. Commun. ACM 16(9), 575\u2013577 (1973)","journal-title":"Commun. ACM"},{"doi-asserted-by":"crossref","unstructured":"Chen, H., Branavan, S., Barzilay, R., Karger, D.R.: Global models of document structure using latent permutations. Association for Computational Linguistics (2009)","key":"11_CR4","DOI":"10.3115\/1620754.1620808"},{"unstructured":"Choi, F.Y.: Advances in domain independent linear text segmentation. In: Proceedings of the 1st North American chapter of the Association for Computational Linguistics Conference, pp. 26\u201333 (2000)","key":"11_CR5"},{"doi-asserted-by":"crossref","unstructured":"Conneau, A., et al.: Unsupervised cross-lingual representation learning at scale. arXiv preprint arXiv:1911.02116 (2019)","key":"11_CR6","DOI":"10.18653\/v1\/2020.acl-main.747"},{"unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186 (2019)","key":"11_CR7"},{"doi-asserted-by":"crossref","unstructured":"Fabricius-Hansen, C.: Information packaging and translation: aspects of translational sentence splitting (German-English\/Norwegian). Sprachspezifische Aspekte der Informationsverteilung pp. 175\u2013214 (1999)","key":"11_CR8","DOI":"10.1515\/9783050078137-008"},{"unstructured":"Feng, F., Yang, Y., Cer, D., Arivazhagan, N., Wang, W.: Language-agnostic Bert sentence embedding. arXiv preprint arXiv:2007.01852 (2020)","key":"11_CR9"},{"unstructured":"Fournier, C.: Evaluating text segmentation using boundary edit distance. In: Proceedings of the 51st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 1702\u20131712 (2013)","key":"11_CR10"},{"doi-asserted-by":"crossref","unstructured":"Glava\u0161, G., Nanni, F., Ponzetto, S.P.: Unsupervised text segmentation using semantic relatedness graphs. In: Proceedings of the Fifth Joint Conference on Lexical and Computational Semantics, pp. 125\u2013130 (2016)","key":"11_CR11","DOI":"10.18653\/v1\/S16-2016"},{"doi-asserted-by":"crossref","unstructured":"Glava\u0161, G., Somasundaran, S.: Two-level transformer and auxiliary coherence modeling for improved text segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 7797\u20137804 (2020)","key":"11_CR12","DOI":"10.1609\/aaai.v34i05.6284"},{"unstructured":"Hearst, M.A.: Texttiling: a quantitative approach to discourse. Technical report USA (1993)","key":"11_CR13"},{"doi-asserted-by":"crossref","unstructured":"Hearst, M.A.: Multi-paragraph segmentation of expository text. In: 32nd Annual Meeting of the Association for Computational Linguistics, pp. 9\u201316 (1994)","key":"11_CR14","DOI":"10.3115\/981732.981734"},{"issue":"1","key":"11_CR15","first-page":"33","volume":"23","author":"MA Hearst","year":"1997","unstructured":"Hearst, M.A.: Text tiling: segmenting text into multi-paragraph subtopic passages. Comput. Linguist. 23(1), 33\u201364 (1997)","journal-title":"Comput. Linguist."},{"issue":"8","key":"11_CR16","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"doi-asserted-by":"crossref","unstructured":"Koehn, P., et al.: Moses: open source toolkit for statistical machine translation. In: Proceedings of the 45th Annual Meeting of the Association for Computational Linguistics Companion Volume Proceedings of the Demo and Poster Sessions, pp. 177\u2013180 (2007)","key":"11_CR17","DOI":"10.3115\/1557769.1557821"},{"doi-asserted-by":"publisher","unstructured":"Koshorek, O., Cohen, A., Mor, N., Rotman, M., Berant, J.: Text segmentation as a supervised learning task. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 2 (Short Papers), pp. 469\u2013473. Association for Computational Linguistics, New Orleans, Louisiana, June 2018. https:\/\/doi.org\/10.18653\/v1\/N18-2075, https:\/\/www.aclweb.org\/anthology\/N18-2075","key":"11_CR18","DOI":"10.18653\/v1\/N18-2075"},{"doi-asserted-by":"crossref","unstructured":"Kozima, H.: Text segmentation based on similarity between words. In: 31st Annual Meeting of the Association for Computational Linguistics, pp. 286\u2013288 (1993)","key":"11_CR19","DOI":"10.3115\/981574.981616"},{"unstructured":"Liu, Y., et al.: Roberta: a robustly optimized Bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)","key":"11_CR20"},{"issue":"1","key":"11_CR21","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1023\/B:INRT.0000009441.78971.be","volume":"7","author":"P McNamee","year":"2004","unstructured":"McNamee, P., Mayfield, J.: Character n-gram tokenization for European language text retrieval. Inf. Retrieval 7(1), 73\u201397 (2004)","journal-title":"Inf. Retrieval"},{"issue":"1","key":"11_CR22","first-page":"21","volume":"17","author":"J Morris","year":"1991","unstructured":"Morris, J., Hirst, G.: Lexical cohesion computed by thesaural relations as an indicator of the structure of text. Comput. Linguist. 17(1), 21\u201348 (1991)","journal-title":"Comput. Linguist."},{"issue":"1","key":"11_CR23","first-page":"103","volume":"23","author":"RJ Passonneau","year":"1997","unstructured":"Passonneau, R.J., Litman, D.J.: Discourse segmentation by human and automated means. Comput. Linguist. 23(1), 103\u2013139 (1997)","journal-title":"Comput. Linguist."},{"issue":"1","key":"11_CR24","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1162\/089120102317341756","volume":"28","author":"L Pevzner","year":"2002","unstructured":"Pevzner, L., Hearst, M.A.: A critique and improvement of an evaluation metric for text segmentation. Comput. Linguist. 28(1), 19\u201336 (2002)","journal-title":"Comput. Linguist."},{"doi-asserted-by":"crossref","unstructured":"Pires, T., Schlinger, E., Garrette, D.: How multilingual is multilingual Bert? arXiv preprint arXiv:1906.01502 (2019)","key":"11_CR25","DOI":"10.18653\/v1\/P19-1493"},{"doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Making monolingual sentence embeddings multilingual using knowledge distillation. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics, November 2020. https:\/\/arxiv.org\/abs\/2004.09813","key":"11_CR26","DOI":"10.18653\/v1\/2020.emnlp-main.365"},{"issue":"2","key":"11_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1149290.1151098","volume":"3","author":"C Sporleder","year":"2006","unstructured":"Sporleder, C., Lapata, M.: Broad coverage paragraph segmentation across languages and domains. ACM Trans. Speech Language Process. (TSLP) 3(2), 1\u201335 (2006)","journal-title":"ACM Trans. Speech Language Process. (TSLP)"},{"doi-asserted-by":"crossref","unstructured":"Utiyama, M., Isahara, H.: A statistical model for domain-independent text segmentation. In: Proceedings of the 39th Annual Meeting of the Association for Computational Linguistics, pp. 499\u2013506 (2001)","key":"11_CR28","DOI":"10.3115\/1073012.1073076"},{"key":"11_CR29","doi-asserted-by":"publisher","DOI":"10.7717\/peerj-cs.1003","volume":"8","author":"P Virameteekul","year":"2022","unstructured":"Virameteekul, P.: Paragraph-level attention based deep model for chapter segmentation. PeerJ Comput. Sci. 8, e1003 (2022)","journal-title":"PeerJ Comput. Sci."}],"container-title":["Communications in Computer and Information Science","Advances in Computational Collective Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-16210-7_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,3,9]],"date-time":"2023-03-09T12:17:06Z","timestamp":1678364226000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-16210-7_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031162091","9783031162107"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-16210-7_11","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"21 September 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}