{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T08:12:35Z","timestamp":1759133555759,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2013,3,18]],"date-time":"2013-03-18T00:00:00Z","timestamp":1363564800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2013,3,18]]},"DOI":"10.1145\/2452376.2452389","type":"proceedings-article","created":{"date-parts":[[2013,3,25]],"date-time":"2013-03-25T14:14:26Z","timestamp":1364220866000},"page":"101-112","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":17,"title":["Computing n-gram statistics in MapReduce"],"prefix":"10.1145","author":[{"given":"Klaus","family":"Berberich","sequence":"first","affiliation":[{"name":"Max Planck Institute for Informatics, Saarbr\u00fccken, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Srikanta","family":"Bedathur","sequence":"additional","affiliation":[{"name":"Indraprastha Institute of Information Technology, New Delhi, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2013,3,18]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Apache Hadoop http:\/\/hadoop.apache.org\/.  Apache Hadoop http:\/\/hadoop.apache.org\/."},{"key":"e_1_3_2_1_2_1","unstructured":"Apache OpenNLP http:\/\/opennlp.apache.org\/.  Apache OpenNLP http:\/\/opennlp.apache.org\/."},{"key":"e_1_3_2_1_3_1","unstructured":"Berkeley DB Java Edition http:\/\/www.oracle.com\/products\/berkeleydb\/.  Berkeley DB Java Edition http:\/\/www.oracle.com\/products\/berkeleydb\/."},{"key":"e_1_3_2_1_4_1","unstructured":"Boilerpipe http:\/\/code.google.com\/p\/boilerpipe\/.  Boilerpipe http:\/\/code.google.com\/p\/boilerpipe\/."},{"key":"e_1_3_2_1_5_1","unstructured":"Google n-Gram Corpus http:\/\/googleresearch.blogspot.de\/2006\/08\/all-our-n-gram-are-belong-to-you.html.  Google n -Gram Corpus http:\/\/googleresearch.blogspot.de\/2006\/08\/all-our-n-gram-are-belong-to-you.html."},{"key":"e_1_3_2_1_6_1","unstructured":"The ClueWeb09 Dataset http:\/\/lemurproject.org\/clueweb09.  The ClueWeb09 Dataset http:\/\/lemurproject.org\/clueweb09."},{"key":"e_1_3_2_1_7_1","unstructured":"The New York Times Annotated Corpus http:\/\/corpus.nytimes.com.  The New York Times Annotated Corpus http:\/\/corpus.nytimes.com."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/170035.170072"},{"key":"e_1_3_2_1_9_1","volume-title":"VLDB","author":"Agrawal R.","year":"1994","unstructured":"R. Agrawal and R. Srikant . Fast algorithms for mining association rules in large databases . VLDB 1994 . R. Agrawal and R. Srikant. Fast algorithms for mining association rules in large databases. VLDB 1994."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/645480.655281"},{"key":"e_1_3_2_1_11_1","volume-title":"The design, implementation, and use of the ngram statistics package. CICLing","author":"Banerjee S.","year":"2003","unstructured":"S. Banerjee and T. Pedersen . The design, implementation, and use of the ngram statistics package. CICLing 2003 . S. Banerjee and T. Pedersen. The design, implementation, and use of the ngram statistics package. CICLing 2003."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.is.2005.11.006"},{"key":"e_1_3_2_1_13_1","volume-title":"Large language models in machine translation. EMNLP-CoNLL","author":"Brants T.","year":"2007","unstructured":"T. Brants Large language models in machine translation. EMNLP-CoNLL 2007 . T. Brants et al. Large language models in machine translation. EMNLP-CoNLL 2007."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1132956.1132958"},{"key":"e_1_3_2_1_15_1","volume-title":"ACL","author":"Ceylan H.","year":"2011","unstructured":"H. Ceylan and R. Mihalcea . An efficient indexer for large n-gram corpora . ACL 2011 . H. Ceylan and R. Mihalcea. An efficient indexer for large n-gram corpora. ACL 2011."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1772690.1772715"},{"key":"e_1_3_2_1_17_1","volume-title":"OSDI","author":"Dean J.","year":"2004","unstructured":"J. Dean and S. Ghemawat . Mapreduce: Simplified data processing on large clusters . OSDI 2004 . J. Dean and S. Ghemawat. Mapreduce: Simplified data processing on large clusters. OSDI 2004."},{"key":"e_1_3_2_1_18_1","volume-title":"INTERSPEECH","author":"Federico M.","year":"2008","unstructured":"M. Federico : an open source toolkit for handling large scale language models . INTERSPEECH 2008 . M. Federico et al. Irstlm: an open source toolkit for handling large scale language models. INTERSPEECH 2008."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2004.03.003"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10618-006-0059-1"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:DAMI.0000005258.31418.83"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-13672-6_3"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1935826.1935857"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1987.1165125"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1718487.1718542"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/1454008.1454027"},{"key":"e_1_3_2_1_27_1","volume-title":"LREC","author":"Lin D.","year":"2010","unstructured":"D. Lin New tools for web-scale n-grams . LREC 2010 . D. Lin et al. New tools for web-scale n-grams. LREC 2010."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1571941.1571970"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/1855013"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/1824795.1824798"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1137\/0222058"},{"key":"e_1_3_2_1_32_1","volume-title":"Science","author":"Michel J.-B.","year":"2010","unstructured":"J.-B. Michel Quantitative Analysis of Culture Using Millions of Digitized Books . Science , 2010 . J.-B. Michel et al. Quantitative Analysis of Culture Using Millions of Digitized Books. Science, 2010."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.14778\/1988776.1988782"},{"key":"e_1_3_2_1_34_1","unstructured":"P. Nguyen etal Msrlm: a scalable language modeling toolkit. MSR-TR-2007-144  P. Nguyen et al. Msrlm: a scalable language modeling toolkit. MSR-TR-2007-144"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/1989323.1989423"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/PL00011656"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2004.77"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/645337.650382"},{"key":"e_1_3_2_1_39_1","volume-title":"INTERSPEECH","author":"Stolcke A.","year":"2002","unstructured":"A. Stolcke . Srilm - an extensible language modeling toolkit . INTERSPEECH 2002 . A. Stolcke. Srilm - an extensible language modeling toolkit. INTERSPEECH 2002."},{"key":"e_1_3_2_1_40_1","volume-title":"An Overview of Microsoft Web N-gram Corpus and Applications. NAACL-HLT","author":"Wang K.","year":"2010","unstructured":"K. Wang An Overview of Microsoft Web N-gram Corpus and Applications. NAACL-HLT 2010 . K. Wang et al. An Overview of Microsoft Web N-gram Corpus and Applications. NAACL-HLT 2010."},{"key":"e_1_3_2_1_41_1","volume-title":"Hadoop: The Definitive Guide","author":"White T.","year":"2010","unstructured":"T. White . Hadoop: The Definitive Guide . O'Reilly Media, Inc. , 2nd edition, 2010 . T. White. Hadoop: The Definitive Guide. O'Reilly Media, Inc., 2nd edition, 2010."},{"key":"e_1_3_2_1_42_1","volume-title":"Managing Gigabytes: Compressing and Indexing Documents and Images","author":"Witten I. H.","year":"1999","unstructured":"I. H. Witten Managing Gigabytes: Compressing and Indexing Documents and Images , Second Edition. Morgan Kaufmann , 1999 . I. H. Witten et al. Managing Gigabytes: Compressing and Indexing Documents and Images, Second Edition. Morgan Kaufmann, 1999."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1162\/089120101300346787"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1006\/jpdc.2000.1695"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007652502315"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000008"}],"event":{"name":"EDBT\/ICDT '13: Joint 2013 EDBT\/ICDT Conferences","acronym":"EDBT\/ICDT '13","location":"Genoa Italy"},"container-title":["Proceedings of the 16th International Conference on Extending Database Technology"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2452376.2452389","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2452376.2452389","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T08:18:24Z","timestamp":1750234704000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2452376.2452389"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,3,18]]},"references-count":46,"alternative-id":["10.1145\/2452376.2452389","10.1145\/2452376"],"URL":"https:\/\/doi.org\/10.1145\/2452376.2452389","relation":{},"subject":[],"published":{"date-parts":[[2013,3,18]]},"assertion":[{"value":"2013-03-18","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}