{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T17:36:44Z","timestamp":1743097004050,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":24,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642201271"},{"type":"electronic","value":"9783642201288"}],"license":[{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-20128-8_5","type":"book-chapter","created":{"date-parts":[[2013,12,13]],"date-time":"2013-12-13T12:15:18Z","timestamp":1386936918000},"page":"93-112","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Methods for Collection and Evaluation of Comparable Documents"],"prefix":"10.1007","author":[{"given":"Monica Lestari","family":"Paramita","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Guthrie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Evangelos","family":"Kanoulas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rob","family":"Gaizauskas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Paul","family":"Clough","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark","family":"Sanderson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,12,14]]},"reference":[{"key":"5_CR1","unstructured":"Adafre, S.F., de Rijke, M.: Finding similar sentences across multiple languages in wikipedia. In: Proceedings of the 11th Conference of the European Chapter of the Association for Computational Linguistics, pp. 62\u201369 (2006)"},{"issue":"3","key":"5_CR2","first-page":"161","volume":"12","author":"D Appelt","year":"1999","unstructured":"Appelt, D.: An introduction to information extraction. Artif. Intell. Commun. 12(3), 161\u2013172 (1999)","journal-title":"Artif. Intell. Commun."},{"key":"5_CR3","unstructured":"Argaw, A.A., Asker, L.: Web mining for an Amharic-English bilingual corpus. In Proceedings of 1st International Conference on Web Information Systems and Technologies (WEBIST 2005), Miami, USA (May 2005)"},{"key":"5_CR4","unstructured":"Baroni, M., Bernardini, S.: Bootstrapping corpora and terms from the web. In Proceedings of LREC (2004)"},{"key":"5_CR5","unstructured":"Cavnar, W.B., Trenkle, J.M.: N-gram-based text categorization. In Proceedings of SDAIR-94, 3rd Annual Symposium on Document Analysis and Information Retrieval, pp. 161\u2013175 (1994)"},{"key":"5_CR6","unstructured":"Chakrabarti, S.: Mining the Web: Discovering Knowledge from Hypertext Data, Science and Technology Books (2002)"},{"issue":"1","key":"5_CR7","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1613\/jair.105","volume":"2","author":"T Dietterich","year":"1995","unstructured":"Dietterich, T., Bakiri, G.: Solving multiclass learning problems via error-correcting output codes. J. Artif. Intell. Res. 2(1), 263\u2013286 (1995)","journal-title":"J. Artif. Intell. Res."},{"key":"5_CR8","doi-asserted-by":"crossref","unstructured":"Do, T., Le, V., Bigi, B., Besacier, L., Castelli, E.: Mining a comparable text corpus for a Vietnamese-French statistical machine translation system. In: Proceedings of the Fourth Workshop on Statistical Machine Translation, pp. 165\u2013172. Association for Computational Linguistics (2009)","DOI":"10.3115\/1626431.1626466"},{"key":"5_CR9","unstructured":"Fung, P., Cheung, P.: Mining very non-parallel corpora: parallel sentence and lexicon extraction vie bootstrapping and EM. In: EMNLP, pp. 57\u201363 (2004)"},{"key":"5_CR10","doi-asserted-by":"crossref","unstructured":"Ghani, R., Jones, R., Mladenic, D.: Building minority language corpora by learning to generate web search queries. KAIS Knowl. Inform. Syst. 7(1) (2005)","DOI":"10.1007\/s10115-003-0121-x"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Grishman, R., Sundheim, B.: Message understanding conference-6: a brief history. In: Proceedings of the 16th International Conference on Computational Linguistics, Copenhagen (June 1996).","DOI":"10.3115\/992628.992709"},{"key":"5_CR12","unstructured":"Hassan, A., Fahmy, H., Hassan, H.: Improving named entity translation by exploiting comparable and parallel corpora. In Proceedings of the 2007 Conference on Recent Advances in Natural Language Processing (RANLP), AMML Workshop (2007)"},{"key":"5_CR13","unstructured":"http:\/\/techcrunch.com\/2010\/02\/24\/twitter-languages\/. Accessed 1 April 2011"},{"key":"5_CR14","doi-asserted-by":"crossref","unstructured":"Mohammadi, M., GhasemAghaee, N.: Building bilingual parallel corpora based on wikipedia. In: Proceedings of Second International Conference on Computer Engineering and Applications, vol. 2, pp. 264\u2013268 (2010)","DOI":"10.1109\/ICCEA.2010.203"},{"issue":"4","key":"5_CR15","doi-asserted-by":"crossref","first-page":"477","DOI":"10.1162\/089120105775299168","volume":"31","author":"D. Munteanu","year":"2005","unstructured":"Munteanu, D., Marcu, D.: Improving machine translation performance by exploiting comparable corpora. Comput. Linguist. 31(4), 477\u2013504 (2005)","journal-title":"Comput. Linguist."},{"key":"5_CR16","unstructured":"Munteanu, D. S., Fraser, A., Marcu, D.: Improved machine translation performance via parallel sentence extraction from comparable corpora. In: HLT-NAACL, pp. 265\u2013272 (2004)"},{"key":"5_CR17","doi-asserted-by":"crossref","unstructured":"Resnik, P.: Mining the web for bilingual text. In: Proceedings of the 37th Annual Meeting of the Association for Computational Linguistics on Computational Linguistics, pp. 527\u2013534, Morristown, NJ, USA. Association for Computational Linguistics (1999)","DOI":"10.3115\/1034678.1034757"},{"key":"5_CR18","volume-title":"Introduction to Modern Information Retrieval","author":"G Salton","year":"1983","unstructured":"Salton, G., McGill, M.J.: Introduction to Modern Information Retrieval. McGraw-Hill, New York (1983)"},{"issue":"11","key":"5_CR19","doi-asserted-by":"publisher","first-page":"613","DOI":"10.1145\/361219.361220","volume":"18","author":"G Salton","year":"1975","unstructured":"Salton, G., Wong, A., Yang, C.S.: A vector space model for automatic indexing. Commun. ACM 18(11), 613\u2013620 (1975)","journal-title":"Commun. ACM"},{"key":"5_CR20","unstructured":"Schonfeld, E.: Costolo: Twitter now has 190 million users tweeting 65 million times a day. (2010). http:\/\/techcrunch.com\/2010\/06\/08\/twitter-190-million-users\/ Accessed 1 September 2010"},{"key":"5_CR21","unstructured":"Sparck-Jones, K., Willet, P.: Readings in Information Retrieval. Morgan Kauffmann, San Francisco (1997)"},{"issue":"4","key":"5_CR22","doi-asserted-by":"publisher","first-page":"257","DOI":"10.2498\/cit.2005.04.01","volume":"13","author":"R Steinberger","year":"2005","unstructured":"Steinberger, R., Pouliquen, B., Ignat, C.: Navigating multilingual news collections using automatically extracted information. J. Comput. Inform. Technol. 13(4), 257\u2013264 (2005)","journal-title":"J. Comput. Inform. Technol."},{"issue":"5","key":"5_CR23","doi-asserted-by":"crossref","first-page":"427","DOI":"10.1007\/s10791-008-9058-8","volume":"11","author":"T. Talvensaari","year":"2008","unstructured":"Talvensaari, T., Pirkola, A., J\u00e4rvelin, K., Juhola, M., Laurikkala, J.: Focused web crawling in the acquisition of comparable corpora. Inform. Retr. 11(5), 427\u2013445 (2008)","journal-title":"Inform. Retr."},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wu, K., Gao, J., Vines, P.: Automatic acquisition of Chinese-English parallel corpus from the web. In: Proceedings of 28th European Conference on Information Retrieval. ECIR \u201906 (2006)","DOI":"10.1007\/11735106_37"}],"container-title":["Building and Using Comparable Corpora"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-20128-8_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,14]],"date-time":"2023-02-14T09:18:53Z","timestamp":1676366333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-642-20128-8_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642201271","9783642201288"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-20128-8_5","relation":{},"subject":[],"published":{"date-parts":[[2013]]},"assertion":[{"value":"14 December 2013","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}