{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T05:10:07Z","timestamp":1746076207039,"version":"3.40.4"},"publisher-location":"Berlin, Heidelberg","reference-count":69,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642201271"},{"type":"electronic","value":"9783642201288"}],"license":[{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-20128-8_3","type":"book-chapter","created":{"date-parts":[[2013,12,13]],"date-time":"2013-12-13T12:15:18Z","timestamp":1386936918000},"page":"51-75","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Automatic Comparable Web Corpora Collection and Bilingual Terminology Extraction for Specialized Dictionary Making"],"prefix":"10.1007","author":[{"given":"Antton","family":"Gurrutxaga","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Igor","family":"Leturia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xabier","family":"Saralegi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"I\u00f1aki San","family":"Vicente","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,12,14]]},"reference":[{"key":"3_CR1","unstructured":"Aduriz, I., Aldezabal, I., Alegria, I., Artola, X., Ezeiza, N., Urizar, R.: Euslem: A lemmatiser\/tagger for basque. In: Proceedings of 7th EURALEX International Conference, vol. 1, pp. 17\u201326. EURALEX, G\u00f6teborg, Sweden (2002)"},{"key":"3_CR2","doi-asserted-by":"crossref","unstructured":"Al-Onaizan, Y., Knight, K.: Machine transliteration of names in arabic text. In: Proceedings of the ACL-02 workshop on Computational approaches to Semitic languages, pp. 1\u201313. ACL, Philadelphia, USA (2002)","DOI":"10.3115\/1118637.1118642"},{"key":"3_CR3","unstructured":"Alegria, I., Gurrutxaga, A., Lizaso, P., Saralegi, X., Ugartetxea, S., Urizar, R.: Linguistic and statistical approaches to basque term extraction. In: Proceedings of GLAT 2004. Barcelona, Spain (2004)"},{"key":"3_CR4","unstructured":"Alegria, I., Gurrutxaga, A., Lizaso, P., Saralegi, X., Ugartetxea, S., Urizar, R.: An xml-based term extraction tool for basque. In: Proceedings of the 4th International Conference on Language Resources and Evaluations (LREC). ELRA, Lisbon, Portugal (2004)"},{"key":"3_CR5","unstructured":"Alegria, I., Gurrutxaga, A., Saralegi, X., Ugartetxea, S.: Elexbi, a basic tool for bilingual term extraction from Spanish-Basque parallel corpora. In: Proceedings of Euralex 2006, pp. 159\u2013165. Euralex, Torino, Italy (2006)"},{"issue":"4","key":"3_CR6","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1145\/582415.582416","volume":"20","author":"G Amati","year":"2002","unstructured":"Amati, G., Van Rijsbergen, C.: Probabilistic models of information retrieval based on measuring divergence from randomness. Trans. Inform. Syst. 20(4), 357\u2013389 (2002)","journal-title":"Trans. Inform. Syst."},{"key":"3_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-010-0844-0","volume-title":"Word Frequency Distributions","author":"R Baayen","year":"2001","unstructured":"Baayen, R.: Word Frequency Distributions. Kluwer, Dordrecht (2001)"},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Ballesteros, L., Croft, W.: Resolving ambiguity for cross-language retrieval. In: Proceedings of SIGIR Conference, pp. 64\u201371. ACM, Melbourne (1998)","DOI":"10.1145\/290941.290958"},{"key":"3_CR9","unstructured":"Baroni, M., Bernardini, S.: Bootcat: Bootstrapping corpora and terms from the web. In: Proceedings of LREC 2004, pp. 1313\u20131316. ELRA, Lisbon, Portugal (2004)"},{"key":"3_CR10","unstructured":"Baroni, M., Chantree, F., Kilgarriff, A., Sharoff, S.: Cleaneval: a competition for cleaning web pages. In: Proceedings of LREC 2008. ELRA, Marrakech, Morocco (2008)"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"Baroni, M., Kilgarriff, A.: Large linguistically-processed web corpora for multiple languages. In: Proceedings of EACL 2006, pp. 87\u201390. EACL, Trento, Italy (2006)","DOI":"10.3115\/1608974.1608976"},{"key":"3_CR12","unstructured":"Baroni, M., Ueyama, M.: Building general- and special purpose corpora by web crawling. In: Proceedings of the 13th NIJL International Symposium. Tokyo, Japan (2006)"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Barzilay, R., Lee, L.: Learning to paraphrase: an unsupervised approach using multiple-sequence alignment. In: Proceedings of HLT\/NAACL, pp. 16\u201323. NAACL, Edmonton, USA (2003)","DOI":"10.3115\/1073445.1073448"},{"key":"3_CR14","unstructured":"Basic dictionary of science and technology, http:\/\/zthiztegia.elhuyar.org"},{"key":"3_CR15","unstructured":"Bekavac, B., Osenova, P., Simov, K., Tadi\u0107, M.: Making monolingual corpora comparable: a case study of Bulgarian & Croatian. In: Proceedings of LREC 2004, pp. 1187\u20131190. ELRA, Lisbon, Portugal (2004)"},{"key":"3_CR16","unstructured":"Blaheta, D., Johnson, M.: Unsupervised learning of multi-word verbs. In: Proceedings of the 39th Annual Meeting of the ACL, pp. 54\u201360. ACL, Toulouse, France (2001)"},{"key":"3_CR17","unstructured":"Bourigault, D.: Lexter, a natural language processing tool for terminology extraction. In: Proceedings of 7th EURALEX International Conference. G\u00f6teborg, Sweden (1996)"},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Braschler, M., Sch\u00e4uble, P.: Multilingual information retrieval based on document alignment techniques. In: Proceedings of the 2nd European Conference on Research and Advanced Technology for Digital Libraries, pp. 183\u2013197. Springer, Heraklion, Greece (1998)","DOI":"10.1007\/3-540-49653-X_12"},{"key":"3_CR19","unstructured":"Broder, A.: On the resemblance and containment of documents. In: Proceedings of Compression and Complexity of Sequences 1997, pp. 21\u201329. IEEE, Salerno, Italy (1997)"},{"key":"3_CR20","doi-asserted-by":"crossref","unstructured":"Broder, A.: Identifying and filtering near-duplicate documents. In: Proceedings of Combinatorial Pattern Matching: 11th Annual Symposium, pp. 1\u201310. Montreal, Canada (2000)","DOI":"10.1007\/3-540-45123-4_1"},{"key":"3_CR21","unstructured":"Cavnar, W., Trenkle, J.: N-gram-based text categorization. In: Proceedings of Third Annual Symposium on Document Analysis and Information Retrieval, pp. 161\u2013175. Las Vegas, USA (1994)"},{"key":"3_CR22","unstructured":"Chakrabarti, S., Van der Berg, M., Dom, B.: Focused crawling: a new approach to topic-specific web resource discovery. In: Proceedings of the 8th International WWW Conference, pp. 545\u2013562. W3C, Toronto, Canada (1999)"},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Chen, H., Bian, G., Lin, W.: Resolving translation ambiguity and target polysemy in cross-language information retrieval. In: Proceedings of the 37th annual meeting of the Association for Computational Linguistics, pp. 215\u2013222. ACL, College Park, USA (1999)","DOI":"10.3115\/1034678.1034717"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Chiao, Y., Zweigenbaum, P.: Looking for candidate translational equivalents in specialized, comparable corpora. In: Proceedings of the 19th International Conference on Computational Linguistics (COLING 2002), pp. 1208\u20131212. ACL, Taipei, Taiwan (2002)","DOI":"10.3115\/1071884.1071904"},{"key":"3_CR25","doi-asserted-by":"crossref","unstructured":"Church, K., Hanks, P.: Word association norms, mutual information and lexicography. In: Proceedings of the 27th Annual Meeting of the ACL, pp. 76\u201383. ACL, Vancouver, Canada (1989)","DOI":"10.3115\/981623.981633"},{"key":"3_CR26","unstructured":"Daille, B.: Combined approach for terminology extraction: lexical statistics and linguistic filtering. Tech. Rep. UCREL Technical Papers 5, UCREL (1995)"},{"key":"3_CR27","doi-asserted-by":"crossref","unstructured":"Daille, B., Morin, E.: French-english terminology extraction from comparable corpora. Natural Language Processing\u2014IJCNLP, p. 707G718 (2005)","DOI":"10.1007\/11562214_62"},{"key":"3_CR28","unstructured":"Dias, G., Guillor\u00e9, S., Lopes, J.: Mutual expectation: a measure for multiword lexical unit extraction. In: Proceedings of VExTAL\u2014Venezia per il Trattamento Automatico delle Lingue. Venezia, Italy (1999)"},{"issue":"1","key":"3_CR29","first-page":"61","volume":"19","author":"T Dunning","year":"1994","unstructured":"Dunning, T.: Accurate methods for the statistics of surprise and coincidence. Comput. Linguist. 19(1), 61\u201374 (1994)","journal-title":"Comput. Linguist."},{"key":"3_CR30","unstructured":"Ferraresi, A., Zanchetta, E., Baroni, M., Bernardini, S.: Introducing and evaluating ukwac, a very large web-derived corpus of English. In: Proceedings of WAC4 Workshop. ACL SIGWAC, Marrakech, Morocco (2008)"},{"key":"3_CR31","unstructured":"Finn, A., Kushmerick, N., Smyth, B.: Fact or fiction: content classification for digital libraries. In: Proceedings of Personalisation and Recommender Systems in Digital Libraries Workshop. Dublin, Ireland (2001)"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Fletcher, W.: Corpus Linguistics in North America 2002. In: Making the Web More Useful as a Source for Linguistic Corpora. Rodopi, Amsterdam (2004)","DOI":"10.1163\/9789004333772_011"},{"key":"3_CR33","unstructured":"Fung, P.: Compiling bilingual lexicon entries from a non-parallel English-Chinese corpus. In: Proceedings of the Third Workshop on Very Large Corpora, pp. 173\u2013183. Boston, USA (1995)"},{"key":"3_CR34","doi-asserted-by":"crossref","unstructured":"Fung, P., Yee, L.: An ir approach for translating new words from nonparallel comparable texts. In: Proceedings of COLING-ACL, pp. 414\u2013420. ACL, Montreal, Canada (1998)","DOI":"10.3115\/980451.980916"},{"key":"3_CR35","unstructured":"Gamallo, P.: Learning bilingual lexicons from comparable English and Spanish corpora. In: Proceedings of Machine Translation Summit XI, pp. 191\u2013198. Copenhagen, Denmark (2007)"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Gao, J., Nie, J.: A study of statistical models for query translation: finding a good unit of translation. In: Proceedings of SIGIR Conference, pp. 194\u2013201. ACM, Seattle, USA (2006)","DOI":"10.1145\/1148170.1148207"},{"key":"3_CR37","unstructured":"Gurrutxaga, A., Leturia, I., Saralegi, X., San Vicente, I.: Evaluation of an automatic process for specialized web corpora collection and term extraction for basque. In: Proceedings of eLexicography 2009. Presses Universitaires de Louvain, Louvain-la-Neuve, Belgium (2009)"},{"key":"3_CR38","doi-asserted-by":"crossref","unstructured":"Hull, D., Grefenstette, G.: Querying across languages: a dictionary-based approach to multilingual information retrieval. In: Proceedings of the 19th annual international ACM SIGIR conference on Research and development in information retrieval, pp. 49\u201357. ACM (1996)","DOI":"10.1145\/243199.243212"},{"key":"3_CR39","unstructured":"Justeson, J.: Technical terminology: Some linguistic properties and an algorithm for identification in text. Tech. Rep. IBM Research Report RC 18906 (82591), IBM (1993)"},{"key":"3_CR40","unstructured":"Kilgarriff, A.: Using word frequency lists to measure corpus homogeneity and similarity between corpora. In: Proceedings of workshop on very large corpora, pp. 231\u2013245. ACL SIGDAT, Beijing and Hong Kong, China (1997)"},{"key":"3_CR41","unstructured":"Kilgarriff, A., Rose, T.: Measures for corpus similarity and homogeneity. In: Proceedings of EMNLP-3, pp. 46\u201352. ACL SIGDAT, Granada, Spain (1998)"},{"key":"3_CR42","unstructured":"Leturia, I., Gurrutxaga, A., Alegria, I., Ezeiza, A.: Corpeus, a \u2019web as corpus\u2019 tool designed for the agglutinative nature of basque. In: Proceedings of the 3rd Web as Corpus Workshop, pp. 69\u201381. Presses Universitaires de Louvain, Louvain-la-Neuve, Belgium (2007)"},{"key":"3_CR43","unstructured":"Leturia, I., Gurrutxaga, A., Alegria, I., Ezeiza, A.: Kimatu, a tool for cleaning non-content text parts from html docs. In: Proceedings of the 3rd Web as Corpus Workshop, pp. 163\u2013167. Presses Universitaires de Louvain, Louvain-la-Neuve, Belgium (2007)"},{"key":"3_CR44","unstructured":"Leturia, I., Gurrutxaga, A., Areta, N., Alegria, I., Ezeiza, A.: Eusbila, a search service designed for the agglutinative nature of basque. In: Proceedings of Improving non-English web searching (iNEWS\u201907) workshop, pp. 47\u201354. SIGIR, Amsterdam, The Netherlands (2007)"},{"key":"3_CR45","unstructured":"Leturia, I., Gurrutxaga, A., Areta, N., Pociello, E.: Analysis and performance of morphological query expansion and language-filtering words on basque web searching. In: Proceedings of LREC 2008. ELRA, Marrakech, Morocco (2008)"},{"key":"3_CR46","unstructured":"Leturia, I., San Vicente, I., Saralegi, X., Lopez de Lacalle, M.: Basque specialized corpora from the web: language-specific performance tweaks and improving topic precision. In: Proceedings of the 4th Web as Corpus Workshop, pp. 40\u201346. ACL SIGWAC, Marrakech, Morocco (2008)"},{"key":"3_CR47","doi-asserted-by":"crossref","unstructured":"Liu, Y., Jin, R., Chai, J.: A maximum coherence model for dictionary-based cross-language information retrieval. In: Proceedings of SIGIR Conference, pp. 536\u2013543. ACM, Salvador, Brazil (2005)","DOI":"10.1145\/1076034.1076125"},{"issue":"3","key":"3_CR48","doi-asserted-by":"crossref","first-page":"217","DOI":"10.1527\/tjsai.17.217","volume":"17","author":"Y Matsuo","year":"2000","unstructured":"Matsuo, Y., Ishizuka, M.: Keyword extraction from a document using word co-occurrence statistical information. Trans. Jpn. Soc. Artif. Intell. 17(3), 217\u2013223 (2000)","journal-title":"Trans. Jpn. Soc. Artif. Intell."},{"key":"3_CR49","unstructured":"Melamed, I.D.: Bitext maps and alignment via pattern recognition. Comput. Linguist. 25(1), 107\u2013130 (1999), http:\/\/portal.acm.org\/citation.cfm?id=973215.973218"},{"key":"3_CR50","unstructured":"Milos, E., Zhang, Y., He, B., Dong, L.: Automatic term extraction and document similarity in special text corpora. In: Proceedings of the Sixth Conference of the Pacific Association for Computational Linguistics, pp. 275\u2013284. Halifax, Canada (2003)"},{"key":"3_CR51","unstructured":"Morin, E., Daille, B., Takeuchi, K., Kageura, K.: Bilingual terminology mining\u2014using brain, not brawn comparable corpora. In: Proceedings of the 45th Annual Meeting of the Association of Computational Linguistics, pp. 664\u2013671. ACL, Prague, Czech Republic (2007)"},{"issue":"4","key":"3_CR52","doi-asserted-by":"publisher","first-page":"477","DOI":"10.1162\/089120105775299168","volume":"31","author":"D Munteanu","year":"2005","unstructured":"Munteanu, D., Marcu, D.: Improving machine translation performance by exploiting non-parallel corpora. Comput. Linguist. 31(4), 477\u2013504 (2005)","journal-title":"Comput. Linguist."},{"key":"3_CR53","doi-asserted-by":"crossref","unstructured":"Pirkola, A.: The effects of query structure and dictionary setups in dictionary-based cross-language information retrieval. In: Proceedings of SIGIR Conference, pp. 55\u201363. ACM, Melbourne, Australia (1998)","DOI":"10.1145\/290941.290957"},{"key":"3_CR54","doi-asserted-by":"crossref","unstructured":"Rapp, R.: Identifying word translations in non-parallel texts. In: Proceedings of the 33rd Annual Meeting of the Association for Computational Linguistics, pp. 320\u2013322. ACL, Cambridge, USA (1995)","DOI":"10.3115\/981658.981709"},{"key":"3_CR55","doi-asserted-by":"crossref","unstructured":"Rapp, R.: Automatic identification of word translations from unrelated English and German corpora. In: Proceedings of the 37th annual meeting of the Association for Computational Linguistics, pp. 519\u2013526. ACL, College Park, USA (1999)","DOI":"10.3115\/1034678.1034756"},{"key":"3_CR56","doi-asserted-by":"crossref","unstructured":"Rayson, P., Garside, R.: Comparing corpora using frequency profiling. In: Proceedings of the Workshop on Comparing Corpora, pp. 1\u20136. ACL, Hong Kong, China (2000)","DOI":"10.3115\/1117729.1117730"},{"key":"3_CR57","doi-asserted-by":"crossref","unstructured":"Robertson, S., Walker, S., Beaulieu, M.: Okapi at trec-7: automatic ad hoc, filtering, vlc and interactive track. In: Proceedings of 7th Text REtrieval Conference (TREC-7), pp. 199\u2013210. Gaithersburg, USA (1998)","DOI":"10.6028\/NIST.SP.500-242.okapi"},{"key":"3_CR58","unstructured":"Saralegi, X., San Vicente, I., Gurrutxaga, A.: Automatic extraction of bilingual terms from comparable corpora in a popular science domain. In: Proceedings of Building and using Comparable Corpora workshop. ACL, Marrakech, Morocco (2008)"},{"key":"3_CR59","first-page":"273","volume":"41","author":"X. Saralegi","year":"2008","unstructured":"Saralegi, X., San Vicente, I., Lopez de Lacalle, M.: Mining term translations from domain restricted comparable corpora. Procesamiento del Lenguaje Natural 41, 273\u2013280 (2008)","journal-title":"Procesamiento del Lenguaje Natural"},{"key":"3_CR60","doi-asserted-by":"crossref","unstructured":"Shao, L., Ng, H.: Mining new word translations from comparable corpora. In: Proceedings of the 20th International Conference on Computational Linguistics (COLING 2004), pp. 618\u2013624. ACL, Geneva, Switzerland (2004)","DOI":"10.3115\/1220355.1220444"},{"key":"3_CR61","unstructured":"Sharoff, S.: WaCky! Working papers on the Web as Corpus, chap. Creating general-purpose corpora using automated search engine queries, pp. 63\u201398. Gedit, Bologna, Italy (2006)"},{"key":"3_CR62","unstructured":"Sharoff, S.: Classifying web corpora into domain and genre using automatic feature identification. In: Proceedings of the 3rd Web as Corpus Workshop, pp. 83\u201394. Presses Universitaires de Louvain, Louvain-la-Neuve, Belgium (2007)"},{"key":"3_CR63","doi-asserted-by":"crossref","unstructured":"Sharoff, S., Babych, B., Hartley, A.: \u2019irrefragable answers\u2019 using comparable corpora to retrieve translation equivalents. Lang. Resour. Eval. 43(1), 15\u201325 (2007), http:\/\/www.springerlink.com\/content\/8k6631431pl3538l\/","DOI":"10.1007\/s10579-007-9046-4"},{"key":"3_CR64","doi-asserted-by":"crossref","unstructured":"Sheridan, P., Ballerini, J.: Experiments in multilingual information retrieval using the spider system. In: Proceedings of the 19th Annual International ACM SIGIR Conference, pp. 58\u201365. ACM, Zurich, Switzerland (1996)","DOI":"10.1145\/243199.243213"},{"issue":"1","key":"3_CR65","first-page":"143","volume":"19","author":"F Smadja","year":"1993","unstructured":"Smadja, F.: Retrieving collocations from text: Xtract. Comput. Linguist. 19(1), 143\u2013177 (1993)","journal-title":"Comput. Linguist."},{"issue":"1","key":"3_CR66","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1145\/1198296.1198300","volume":"25","author":"T Talvensaari","year":"2007","unstructured":"Talvensaari, T., Laurikkala, J., J\u00e4rvelin, K., Juhola, M., Keskustalo, H.: Creating and exploiting a comparable corpus in cross-language information retrieval. ACM Trans. Inform. Syst. 25(1), 4 (2007)","journal-title":"ACM Trans. Inform. Syst."},{"key":"3_CR67","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1007\/s10791-008-9058-8","volume":"11","author":"T Talvensaari","year":"2008","unstructured":"Talvensaari, T., Pirkola, A., J\u00e4rvelin, K., Juhola, M., Laurikkala, J.: Focused web crawling in acquisition of comparable corpora. Inform. Retr. 11, 427\u2013445 (2008)","journal-title":"Inform. Retr."},{"key":"3_CR68","unstructured":"Treetagger, http:\/\/www.ims.uni-stuttgart.de\/projekte\/corplex\/TreeTagger\/"},{"key":"3_CR69","unstructured":"Zientzia.net, http:\/\/www.zientzia.net"}],"container-title":["Building and Using Comparable Corpora"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-20128-8_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T04:49:56Z","timestamp":1746074996000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-642-20128-8_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642201271","9783642201288"],"references-count":69,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-20128-8_3","relation":{},"subject":[],"published":{"date-parts":[[2013]]},"assertion":[{"value":"14 December 2013","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}