{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T23:17:43Z","timestamp":1777331863575,"version":"3.51.4"},"publisher-location":"Cham","reference-count":58,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319990033","type":"print"},{"value":"9783319990040","type":"electronic"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-319-99004-0_3","type":"book-chapter","created":{"date-parts":[[2019,2,6]],"date-time":"2019-02-06T12:11:22Z","timestamp":1549455082000},"page":"55-87","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Collecting Comparable Corpora"],"prefix":"10.1007","author":[{"given":"Monica Lestari","family":"Paramita","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ahmet","family":"Aker","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Paul","family":"Clough","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Robert","family":"Gaizauskas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikos","family":"Glaros","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikos","family":"Mastropavlos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Olga","family":"Yannoutsou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Radu","family":"Ion","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"\u0218tef\u0103nescu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexandru","family":"Ceau\u015fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Tufi\u0219","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Judita","family":"Preiss","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,2,7]]},"reference":[{"key":"3_CR1","unstructured":"ACCURAT Deliverable: D3.3, D3.4, D3.5."},{"key":"3_CR2","unstructured":"Adafre, S. F., & de Rijke, M. (2006). Finding similar sentences across multiple languages in Wikipedia. Proceedings of the EACL Workshop on New Text, Trento, Italy."},{"key":"3_CR3","unstructured":"Aker, A., Kanoulas, E., & Gaizauskas, R. (2012). A light way to collect comparable corpora from the Web. Proceedings of LREC 2012, 21\u201327 May, Istanbul, Turkey."},{"key":"3_CR4","unstructured":"Ard\u00f6, A., & Golub, K. (2007). Documentation for the Combine (Focused) Crawling System. \n                  http:\/\/combine.it.lth.se\/documentation\/DocMain\/"},{"key":"3_CR5","unstructured":"Argaw, A. A., & Asker, L. (2005). Web mining for an amharic-english bilingual corpus. Proceedings of the 1st International Conference on Web Information Systems and Technologies, WEBIST \u201905 (pp. 239\u2013246). INSTICC Press."},{"key":"3_CR6","unstructured":"Baroni, M., & Bernardini, S. (2004). BootCaT: Bootstrapping corpora and terms from the Web. Proceedings of LREC 2004 (pp. 1313\u20131316)."},{"key":"3_CR7","unstructured":"Barzilay, R., & McKeown, K. R. (2001). Extracting paraphrases from a parallel corpus. ACL \u201901: Proceedings of the 39th Annual Meeting on Association for Computational Linguistics (pp. 50\u201357). Association for Computational Linguistics, Morristown, NJ."},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Bharadwaj, R. G., & Varma, V. (2011). Language independent identification of parallel sentences using Wikipedia. Proceedings of the 20th International Conference Companion on World Wide Web, WWW \u201911 (pp. 11\u201312), ACM, New York, NY.","DOI":"10.1145\/1963192.1963199"},{"key":"3_CR9","first-page":"993","volume":"3","author":"DM Blei","year":"2003","unstructured":"Blei, D. M., Ng, A. Y., & Jordan, M. I. (2003). Latent dirichlet allocation. The Journal of Machine Learning Research, 3, 993\u20131022.","journal-title":"The Journal of Machine Learning Research"},{"key":"3_CR10","unstructured":"Braschler, P. S. (1998). Multilingual information retrieval based on document alignment techniques. Research and Advanced Technology for Digital Libraries: Second European Conference, ECDL\u201998, Heraklion, Crete, Cyprus, September 21\u201323, 1998: Proceedings, 183. Springer."},{"issue":"1\u20137","key":"3_CR11","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1016\/S0169-7552(98)00110-X","volume":"30","author":"S Brin","year":"1998","unstructured":"Brin, S., & Page, L. (1998). The anatomy of a large-scale hypertextual Web search engine. Computer Networks and ISDN Systems, 30(1\u20137), 107\u2013117.","journal-title":"Computer Networks and ISDN Systems"},{"key":"3_CR12","doi-asserted-by":"crossref","unstructured":"Callison-Burch, C., Koehn, P., & Osborne, M. (2006). Improved statistical machine translation using paraphrases. Proceedings of the Main Conference on Human Language Technology Conference of the North American Chapter of the Association of Computational Linguistics (pp. 17\u201324). Association for Computational Linguistics, Morristown, NJ.","DOI":"10.3115\/1220835.1220838"},{"key":"3_CR13","unstructured":"Cavnar, W. B., & Trenkle, J. M. (1994). N-gram-based text categorization. Ann Arbor MI, 48113(2), 161\u2013175."},{"key":"3_CR14","doi-asserted-by":"crossref","unstructured":"Chakrabarti, S., Punera, K., & Subramanyam, M. (2002, May). Accelerated focused crawling through online relevance feedback. Proceedings of the 11th International Conference on World Wide Web (pp. 148\u2013159). ACM.","DOI":"10.1145\/511446.511466"},{"issue":"1\u20137","key":"3_CR15","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1016\/S0169-7552(98)00108-1","volume":"30","author":"J Cho","year":"1998","unstructured":"Cho, J., Garcia-Molina, H., & Page, L. (1998). Efficient crawling through URL ordering. Computer Networks and ISDN Systems, 30(1\u20137), 161\u2013172.","journal-title":"Computer Networks and ISDN Systems"},{"issue":"2","key":"3_CR16","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1016\/0169-7552(94)90132-5","volume":"27","author":"PME Bra De","year":"1994","unstructured":"De Bra, P. M. E., & Post, R. D. J. (1994). Information retrieval in the World-Wide Web: Making client-based searching feasible. Computer Networks and ISDN Systems, 27(2), 183\u2013192.","journal-title":"Computer Networks and ISDN Systems"},{"key":"3_CR17","unstructured":"Dimalen, D. M. D., & Roxas, R. (2007). AutoCor: A query based automatic acquisition of corpora of closely-related languages. Proceedings of the 21st PACLIC (pp. 146\u2013154)."},{"key":"3_CR18","doi-asserted-by":"publisher","first-page":"77","DOI":"10.2478\/v10108-010-0003-9","volume":"93","author":"M Espl\u00e0-Gomis","year":"2010","unstructured":"Espl\u00e0-Gomis, M., & Forcada, M. L. (2010). Combining content-based and URL-based heuristics to harvest aligned bitexts from multilingual sites with bitextor. The Prague Bulletin of Mathematical Linguistics, 93, 77\u201386.","journal-title":"The Prague Bulletin of Mathematical Linguistics"},{"key":"3_CR19","unstructured":"Filatova, E. (2009). Directions for exploiting asymmetries in multilingual Wikipedia. Proceedings of the Third International Workshop on Cross Lingual Information Access: Addressing the Information Need of Multilingual Societies (CLIAWS3 \u201909)."},{"key":"3_CR20","unstructured":"Fung, P., & Cheung, P. (2004). Mining very-non-parallel corpora: Parallel sentence and lexicon extraction via bootstrapping and em. Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing, EMNLP \u201904 (pp. 57\u201363), Citeseer."},{"key":"3_CR21","doi-asserted-by":"crossref","unstructured":"Gamallo, P., & Garcia, M. (2012). Extraction of bilingual cognates from Wikipedia. Computational Processing of the Portuguese Language (pp. 63\u201372). Springer.","DOI":"10.1007\/978-3-642-28885-2_7"},{"issue":"1","key":"3_CR22","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1007\/s10115-003-0121-x","volume":"7","author":"R Ghani","year":"2005","unstructured":"Ghani, R., Jones, R., & Mladenic, D. (2005). Building minority language corpora by learning to generate web search queries. Knowledge and Information Systems, 7(1), 56\u201383.","journal-title":"Knowledge and Information Systems"},{"key":"3_CR23","unstructured":"Hassan, A., Fahmy, H., & Hassan, H. (2007). Improving named entity translation by exploiting comparable and parallel corpora. Proceedings of the 2007 Conference on Recent Advances in Natural Language Processing (RANLP), AMML Workshop."},{"issue":"1\u20137","key":"3_CR24","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1016\/S0169-7552(98)00038-5","volume":"30","author":"M Hersovici","year":"1998","unstructured":"Hersovici, M., Jacovi, M., Maarek, Y. S., Pelleg, D., Shtalhaim, M., & Ur, S. (1998). The sharksearch algorithm\u2014An application: Tailored Web site mapping. Computer Networks and ISDN Systems, 30(1\u20137), 317\u2013326.","journal-title":"Computer Networks and ISDN Systems"},{"key":"3_CR25","unstructured":"Huang, D., Zhao, L., Li, L., & Yu, H. (2010). Mining large-scale comparable corpora from Chinese-English news collections. Proceedings of the 23rd International Conference on Computational Linguistics: Posters (pp. 472\u2013480). Association for Computational Linguistics."},{"key":"3_CR26","unstructured":"Ion, R., Tufi\u015f, D., Boro\u015f, T., Ceau\u015fu, A., & \u015etef\u0103nescu, D. (2010). On-line compilation of comparable corpora and their evaluation. Proceedings of the 7th International Conference Formal Approaches to South Slavic and Balkan Languages (FASSBL7) (pp. 29\u201334). Croatian Language Technologies Society \u2013 Faculty of Humanities and Social Sciences, University of Zagreb, Dubrovnik, Croatia, October 2010."},{"key":"3_CR27","doi-asserted-by":"crossref","unstructured":"Kauchak, D., & Barzilay, R. (2006). Paraphrasing for automatic evaluation. Proceedings of the Main Conference on Human Language Technology Conference of the North American Chapter of the Association of Computational Linguistics (pp. 455\u2013462). Association for Computational Linguistics, Morristown, NJ.","DOI":"10.3115\/1220835.1220893"},{"key":"3_CR28","doi-asserted-by":"crossref","unstructured":"Koehn, P. (2009). Statistical machine translation. Cambridge University Press.","DOI":"10.1017\/CBO9780511815829"},{"key":"3_CR29","doi-asserted-by":"crossref","unstructured":"Kohlsch\u00fctter, C., Fankhauser, P., & Nejdl, W. (2010). Boilerplate detection using shallow text features. The Third ACM International Conference on Web Search and Data Mining.","DOI":"10.1145\/1718487.1718542"},{"key":"3_CR30","unstructured":"Kumano, T., Tanaka, H., & Tokunaga, T. (2007). Extracting phrasal alignments from comparable corpora by using joint probability SMT model. Proceedings of the 11th International Conference on Theoretical and Methodological Issues in Machine Translation (TMI-07) (pp. 95\u2013103)."},{"key":"3_CR31","unstructured":"L\u00fc, Y., Huang, J., & Liu, Q. (2007, June). Improving statistical machine translation performance by training data selection and optimization. EMNLP-CoNLL (Vol. 34, pp. 3\u2013350)."},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Marton, Y., Callison-Burch, C., Resnik, P. (2009). Improved statistical machine translation using monolingually-derived paraphrases. Proceedings of the 2009 Conference on Empirical Methods in Natural Language Processing (pp. 381\u2013390). Association for Computational Linguistics.","DOI":"10.3115\/1699510.1699560"},{"key":"3_CR33","unstructured":"Mastropavlos, N., & Papavassiliou, V. (2011). Automatic acquisition of bilingual language resources. Proceedings of the 10th International Conference on Greek Linguistics, Komotini, Greece"},{"issue":"2\u20133","key":"3_CR34","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1023\/A:1007653114902","volume":"39","author":"F Menczer","year":"2000","unstructured":"Menczer, F., & Belew, R. (2000). Adaptive retrieval agents: Internalizing local context and scaling up to the Web. Machine Learning, 39(2\u20133), 203\u2013242.","journal-title":"Machine Learning"},{"key":"3_CR35","unstructured":"Munteanu, D. S., & Marcu, D. (2002). Processing comparable corpora with bilingual suffix trees. EMNLP \u201902: Proceedings of the ACL-02 Conference on Empirical Methods in Natural Language Processing (pp. 289\u2013295). Association for Computational Linguistics, Morristown, NJ."},{"issue":"4","key":"3_CR36","doi-asserted-by":"publisher","first-page":"477","DOI":"10.1162\/089120105775299168","volume":"31","author":"DS Munteanu","year":"2005","unstructured":"Munteanu, D. S., & Marcu, D. (2005). Improving machine translation performance by exploiting non-parallel corpora. Computational Linguistics, 31(4), 477\u2013504.","journal-title":"Computational Linguistics"},{"key":"3_CR37","unstructured":"Munteanu, D. S., & Marcu, D. (2006). Extracting parallel sub-sentential fragments from non-parallel corpora. ACL-44: Proceedings of the 21st International Conference on Computational Linguistics and the 44th Annual Meeting of the Association for Computational Linguistics (pp. 81\u201388). Association for Computational Linguistics, Morristown, NJ."},{"key":"3_CR38","unstructured":"Nakov, P. (2008). Paraphrasing verbs for noun compound interpretation. Proceedings of the Workshop on Multiword Expressions, LREC-2008."},{"key":"3_CR39","unstructured":"Paramita, M., Clough, P., Aker, A., & Gaizauskas, R. (2012). Correlation between similarity measures for inter-language linked Wikipedia articles. Proceedings of the Eighth International Conference on Language Resources and Evaluation (LREC 2012) (pp. 790\u2013797), Istanbul, Turkey."},{"key":"3_CR40","doi-asserted-by":"publisher","first-page":"33","DOI":"10.1007\/3-540-45411-X_4","volume-title":"AI*IA 2001: Advances in Artificial Intelligence","author":"Andrea Passerini","year":"2001","unstructured":"Passerini, A., Frasconi, P., & Soda, G. (2001). Evaluation methods for focused crawling, Lecture Notes in Computer Science 2175, pp. 33\u201345."},{"key":"3_CR41","doi-asserted-by":"crossref","unstructured":"Phan, X. H., Nguyen, L. M., & Horiguchi, S. (2008, April). Learning to classify short and sparse text and web with hidden topics from large-scale data collections. Proceedings of the 17th International Conference on World Wide Web (pp. 91\u2013100). ACM.","DOI":"10.1145\/1367497.1367510"},{"key":"3_CR42","unstructured":"Pinkerton, B. (1994). Finding what people want: Experiences with the Web Crawler. Proceedings of the 2nd International World Wide Web Conference."},{"key":"3_CR43","unstructured":"Preiss, J. (2012). Identifying comparable corpora using LDA. Proceedings of the 2012 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL HLT \u201812) (pp. 558\u2013562). Association for Computational Linguistics, Stroudsburg, PA."},{"key":"3_CR44","doi-asserted-by":"crossref","unstructured":"Rapp, R. (1999). Automatic identification of word translations from unrelated English and German corpora. Proceedings of the 37th Annual Meeting of the Association for Computational Linguistics on Computational Linguistics (pp. 519\u2013526). Association for Computational Linguistics.","DOI":"10.3115\/1034678.1034756"},{"key":"3_CR45","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/3-540-49478-2_7","volume-title":"Machine Translation and the Information Soup","author":"Philip Resnik","year":"1998","unstructured":"Resnik, P. (1998). Parallel strands: A preliminary investigation into mining the web for bilingual text. In D. Farwell, L. Gerber, & E. Hovy (Eds.), Machine Translation and the Information Soup: Third Conference of the Association for Machine Translation in the Americas (AMTA-98), Langhorne, PA, Lecture Notes in Artificial Intelligence 1529, Springer, October, 1998."},{"key":"3_CR46","doi-asserted-by":"crossref","unstructured":"Resnik, P. (1999). Mining the web for bilingual text. Proceedings of the 37th Annual Meeting of the Association for Computational Linguistics on Computational Linguistics (pp. 527\u2013534). Association for Computational Linguistics.","DOI":"10.3115\/1034678.1034757"},{"key":"3_CR47","unstructured":"Rose, T. G., Stevenson, M., & Whitehead, M. (2002). The Reuters corpus volume 1 \u2013 from yesterday\u2019s news to tomorrow\u2019s language resources. Proceedings of the Third International Conference on Language Resources and Evaluation (pp. 827\u2013832)."},{"key":"3_CR48","doi-asserted-by":"crossref","unstructured":"Sharoff, S., Babych, B., & Hartley, A. (2006). Using comparable corpora to solve problems difficult for human translators. Proceedings of the COLING\/ACL on Main Conference Poster Sessions (pp. 739\u2013746). Association for Computational Linguistics, Morristown, NJ.","DOI":"10.3115\/1273073.1273168"},{"key":"3_CR49","unstructured":"Simard, M., Foster, G. F., & Isabelle, P. (1993). Using cognates to align sentences in bilingual corpora. In A. Gawman, E. Kidd, & P-\u00c5. Larson (Eds.), Proceedings of the 1993 Conference of the Centre for Advanced Studies on Collaborative Research: Distributed Computing (CASCON \u201993) (Vol. 2, pp. 1071\u20131082). IBM Press."},{"key":"3_CR58","first-page":"403","volume-title":"NAACL-HLT","author":"JR Smith","year":"2010","unstructured":"Smith, J. R., Quirk, C., & Toutanova, K. (2010). Extracting parallel sentences from comparable corpora using document level alignment. In NAACL-HLT (pp. 403\u2013411)."},{"issue":"4","key":"3_CR50","doi-asserted-by":"publisher","first-page":"257","DOI":"10.2498\/cit.2005.04.01","volume":"13","author":"R Steinberger","year":"2005","unstructured":"Steinberger, R., Pouliquen, B., & Ignat, C. (2005). Navigating multilingual news collections using automatically extracted information. Journal of Computing and Information Technology, 13(4), 257\u2013264.","journal-title":"Journal of Computing and Information Technology"},{"issue":"5","key":"3_CR51","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1007\/s10791-008-9058-8","volume":"11","author":"T Talvensaari","year":"2008","unstructured":"Talvensaari, T., Pirkola, A., J\u00e4rvelin, K., Juhola, M., & Laurikkala, J. (2008). Focused web crawling in the acquisition of comparable corpora. Information Retrieval, 11(5), 427\u2013445.","journal-title":"Information Retrieval"},{"key":"3_CR52","doi-asserted-by":"crossref","unstructured":"Theobald, M., Siddharth, J., & Paepcke, A. (2008). SpotSigs: Robust and efficient near duplicate detection in large web collections. 31st Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2008).","DOI":"10.1145\/1390334.1390431"},{"key":"3_CR53","unstructured":"Tom\u00e1s, J., Bataller, J., Casacuberta, F., & Lloret, J., (2001). Mining Wikipedia as a parallel and comparable corpus. Language Forum (Vol. 34, No. 1, pp. 123\u2013137). Bahri Publications."},{"key":"3_CR54","unstructured":"Uszkoreit, J., Ponte, J. M., Popat, A. C., & Dubiner, M. (2010, August). Large scale parallel document mining for machine translation. Proceedings of the 23rd International Conference on Computational Linguistics (pp. 1101\u20131109). Association for Computational Linguistics."},{"key":"3_CR55","unstructured":"Yu, K., & Tsujii, J. (2009). Extracting bilingual dictionary from comparable corpora with dependency heterogeneity. Proceedings of Human Language Technologies: The 2009 Annual Conference of the North American Chapter of the Association for Computational Linguistics, Companion Volume: Short Papers (pp. 121\u2013124). Association for Computational Linguistics, Stroudsburg, PA."},{"key":"3_CR56","unstructured":"Zhao, S., Niu, C., Zhou, M., Liu, T., & Li, S. (2008, June). Combining multiple resources to improve SMT-based paraphrasing model. Proceedings of ACL-08: HLT (pp. 1021\u20131029). Association for Computational Linguistics, Columbus, OH."},{"key":"3_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wu, K., Gao, J., & Vines, P. (2006). Automatic acquisition of Chinese-English parallel corpus from the web. Proceedings of 28th European Conference on Information Retrieval ECIR 2006, April 10\u201312, 2006, London.","DOI":"10.1007\/11735106_37"}],"container-title":["Theory and Applications of Natural Language Processing","Using Comparable Corpora for Under-Resourced Areas of Machine Translation"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-99004-0_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,21]],"date-time":"2019-05-21T04:24:28Z","timestamp":1558412668000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-99004-0_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783319990033","9783319990040"],"references-count":58,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-99004-0_3","relation":{},"ISSN":["2192-032X","2192-0338"],"issn-type":[{"value":"2192-032X","type":"print"},{"value":"2192-0338","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"7 February 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}