{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T21:43:05Z","timestamp":1770414185904,"version":"3.49.0"},"publisher-location":"Berlin, Heidelberg","reference-count":25,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783642450679","type":"print"},{"value":"9783642450686","type":"electronic"}],"license":[{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-45068-6_18","type":"book-chapter","created":{"date-parts":[[2013,12,9]],"date-time":"2013-12-09T12:55:05Z","timestamp":1386593705000},"page":"203-214","source":"Crossref","is-referenced-by-count":9,"title":["Duplicate News Story Detection Revisited"],"prefix":"10.1007","author":[{"given":"Omar","family":"Alonso","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dennis","family":"Fetterly","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark","family":"Manasse","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"18_CR1","unstructured":"Alonso, O.: Implementing crowdsourcing-based relevance experimentation: An industrial perspective. In: Information Retrieval, pp. 1\u201320 (2012)"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Bendersky, M., Croft, W.B.: Finding text reuse on the web. In: WSDM, pp. 262\u2013271 (2009)","DOI":"10.1145\/1498759.1498835"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Broder, A.Z., Glassman, S.C., Manasse, M.S., Zweig, G.: Syntactic clustering of the web. In: WWW, pp. 1157\u20131166 (1997)","DOI":"10.1016\/S0169-7552(97)00031-7"},{"key":"18_CR4","unstructured":"Buckley, C., Salton, G., Allan, J.: Automatic retrieval with locality information using SMART. In: TREC-1, pp. 69\u201372 (1992)"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Charikar, M.S.: Similarity estimation techniques from rounding algorithms. In: ACM STOC, pp. 380\u2013388 (2002)","DOI":"10.1145\/509907.509965"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Chowdhury, A., Frieder, O., Grossman, D., McCabe, M.C.: Collection statistics for fast duplicate document detection. ACM TOIS\u00a020(2) (2002)","DOI":"10.1145\/506309.506311"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Fetterly, D., Manasse, M., Najork, M.: Detecting phrase-level duplication on the world wide web. In: ACM SIGIR, pp. 170\u2013177 (2005)","DOI":"10.1145\/1076034.1076066"},{"key":"18_CR8","unstructured":"Gibson, J., Wellner, B., Lubar, S.: Identification of duplicate news stories in web pages. In: Proceedings of the 4th Web as CorpusWorkshop, WAC-4 (2008)"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Gollapudi, S., Panigrahy, R.: Exploiting asymmetry in hierarchical topic extraction. In: ACM CIKM, pp. 475\u2013482 (2006)","DOI":"10.1145\/1183614.1183683"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Henzinger, M.: Finding near-duplicate web pages: A large-scale evaluation of algorithms. In: ACM SIGIR, pp. 284\u2013291 (2006)","DOI":"10.1145\/1148170.1148222"},{"issue":"3","key":"18_CR11","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1002\/asi.10170","volume":"54","author":"T.C. Hoad","year":"2003","unstructured":"Hoad, T.C., Zobel, J.: Methods for identifying versioned and plagiarized documents. J.\u00a0Am. Soc. Inf. Sci. Technol.\u00a054(3), 203\u2013215 (2003)","journal-title":"J.\u00a0Am. Soc. Inf. Sci. Technol."},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Ioffe, S.: Improved consistent sampling, weighted minhash and \u21131 sketching. In: IEEE ICDM, pp. 246\u2013255 (2010)","DOI":"10.1109\/ICDM.2010.80"},{"key":"18_CR13","doi-asserted-by":"crossref","unstructured":"Kienreich, W., Granitzer, M., Sabol, V., Klieber, W.: Plagiarism detection in large sets of press agency news articles. In: Database and Expert Systems Applications, pp. 181\u2013188 (2006)","DOI":"10.1109\/DEXA.2006.112"},{"key":"18_CR14","unstructured":"Manasse, M., McSherry, F., Talwar, K.: Consistent weighted sampling. Technical Report MSR-TR-2010-73, Microsoft Research (2010)"},{"key":"18_CR15","unstructured":"Manber, U.: Finding similar files in a large file system. In: USENIX WTEC, Berkeley (1994)"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Manku, G.S., Jain, A., Das Sarma, A.: Detecting near-duplicates for web crawling. In: WWW, pp. 141\u2013150 (2007)","DOI":"10.1145\/1242572.1242592"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Muthitacharoen, A., Chen, B., Mazi\u00e8res, D.: A low-bandwidth network file system. In: ACM SOSP, pp. 174\u2013187 (2001)","DOI":"10.1145\/502059.502052"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Najork, M.: Detecting quilted web pages at scale. In: ACM SIGIR (2012)","DOI":"10.1145\/2348283.2348337"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Pasternack, J., Roth, D.: Extracting article text from the web with maximum subsequence segmentation. In: WWW, pp. 971\u2013980 (2009)","DOI":"10.1145\/1526709.1526840"},{"key":"18_CR20","unstructured":"Patel, R.: UHRS overview, http:\/\/research.microsoft.com\/en-us\/um\/redmond\/events\/fs2012\/presentations\/Rajesh_Patel.pdf"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Schleimer, S., Wilkerson, D.S., Aiken, A.: Winnowing: Local algorithms for document fingerprinting. In: ACM SIGMOD, pp. 76\u201385 (2003)","DOI":"10.1145\/872757.872770"},{"key":"18_CR22","doi-asserted-by":"crossref","unstructured":"Stein, B., zu Eissen, S.M., Potthast, M.: Strategies for retrieving plagiarized documents. In: ACM SIGIR, pp. 825\u2013826 (2007)","DOI":"10.1145\/1277741.1277928"},{"key":"18_CR23","unstructured":"Teodosiu, D., Bj\u00f8rner, N., Gurevich, Y., Manasse, M., Porkka, J.: Optimizing file replication over limited-bandwidth networks using remote differential compression. Technical Report MSR-TR-2006-157, Microsoft Research (2006)"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Theobald, M., Siddharth, J., Paepcke, A.: Spotsigs: robust and efficient near duplicate detection in large web collections. In: ACM SIGIR, pp. 563\u2013570 (2008)","DOI":"10.1145\/1390334.1390431"},{"key":"18_CR25","unstructured":"Tridgell, A., Mackerras, P.: The rsync algorithm. Technical Report TR-CS-96-05, Australian National University, Dept. of Computer Science (June 1996), http:\/\/rsync.samba.org"}],"container-title":["Lecture Notes in Computer Science","Information Retrieval Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-45068-6_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T03:31:49Z","timestamp":1746070309000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-642-45068-6_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642450679","9783642450686"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-45068-6_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013]]}}}