{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T06:06:15Z","timestamp":1777442775403,"version":"3.51.4"},"publisher-location":"Berlin, Heidelberg","reference-count":24,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783642047688","type":"print"},{"value":"9783642047695","type":"electronic"}],"license":[{"start":{"date-parts":[[2009,1,1]],"date-time":"2009-01-01T00:00:00Z","timestamp":1230768000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2009]]},"DOI":"10.1007\/978-3-642-04769-5_18","type":"book-chapter","created":{"date-parts":[[2009,9,28]],"date-time":"2009-09-28T12:04:25Z","timestamp":1254139465000},"page":"205-217","source":"Crossref","is-referenced-by-count":8,"title":["Exploiting Sentence-Level Features for Near-Duplicate Document Detection"],"prefix":"10.1007","author":[{"given":"Jenq-Haur","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hung-Chi","family":"Chang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"7","key":"18_CR1","doi-asserted-by":"publisher","first-page":"595","DOI":"10.1016\/j.is.2005.11.006","volume":"31","author":"Y. Bernstein","year":"2006","unstructured":"Bernstein, Y., Zobel, J.: Accurate Discovery of Co-derivative Documents via Duplicate Text Detection. Information Systems\u00a031(7), 595\u2013609 (2006)","journal-title":"Information Systems"},{"key":"18_CR2","doi-asserted-by":"crossref","unstructured":"Brin, S., Davis, J., Garcia-Molina, H.: Copy Detection Mechanisms for Digital Documents. In: The 1995 ACM International Conference on Management of Data (SIGMOD 1995), pp. 398\u2013409 (1995)","DOI":"10.1145\/223784.223855"},{"key":"18_CR3","unstructured":"Broder, A.: On the Resemblance and Containment of Documents. In: Compression and Complexity of Sequences, pp. 21\u201329 (1997)"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Broder, A., Glassman, S., Manasse, M., Zweig, G.: Syntactic Clustering of the Web. In: The 6th International Conference on World Wide Web (WWW 1997), pp. 393\u2013404 (1997)","DOI":"10.1016\/S0169-7552(97)00031-7"},{"key":"18_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"410","DOI":"10.1007\/978-3-540-77094-7_52","volume-title":"Asian Digital Libraries. Looking Back 10 Years and Forging New Frontiers","author":"H.C. Chang","year":"2007","unstructured":"Chang, H.C., Wang, J.H.: Organizing News Archives by Near-duplicate Copy Detection in Digital Libraries. In: Goh, D.H.-L., Cao, T.H., S\u00f8lvberg, I.T., Rasmussen, E. (eds.) ICADL 2007. LNCS, vol.\u00a04822, pp. 410\u2013419. Springer, Heidelberg (2007)"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Chang, H.C., Wang, J.H., Chiu, C.Y.: Finding Event-Relevant Content from the Web Using a Near-duplicate Detection Approach. In: The 2007 IEEE\/WIC\/ACM International Conference on Web Intelligence (WI 2007), pp. 291\u2013294 (2007)","DOI":"10.1109\/WI.2007.25"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Charikar, M.S.: Similarity Estimation Techniques from Rounding Algorithms. In: The 34th Annual ACM Symposium on Theory of Computing (STOC 2002), pp. 380\u2013388 (2002)","DOI":"10.1145\/509907.509965"},{"issue":"2","key":"18_CR8","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1145\/506309.506311","volume":"20","author":"A. Chowdhury","year":"2002","unstructured":"Chowdhury, A., Frieder, O., Grossman, D., McCabe, M.C.: Collection Statistics for Fast Duplicate Document Detection. ACM Transactions on Information Systems (TOIS)\u00a020(2), 171\u2013191 (2002)","journal-title":"ACM Transactions on Information Systems (TOIS)"},{"issue":"3","key":"18_CR9","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1145\/363958.363994","volume":"7","author":"F.J. Damerau","year":"1964","unstructured":"Damerau, F.J.: A Technique for Computer Detection and Correction of Spelling Errors. Communications of the ACM\u00a07(3), 171\u2013176 (1964)","journal-title":"Communications of the ACM"},{"key":"18_CR10","doi-asserted-by":"crossref","unstructured":"Fetterly, D., Manasse, M., Najork, M.: Detecting Phrase-level Duplication on the World Wide Web. In: The 28th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2005), pp. 170\u2013177 (2005)","DOI":"10.1145\/1076034.1076066"},{"key":"18_CR11","unstructured":"Heintze, N.: Scalable Document Fingerprinting. In: The 2nd USENIX Workshop on Electronic Commerce (1996)"},{"key":"18_CR12","doi-asserted-by":"crossref","unstructured":"Henzinger, M.: Finding Near-duplicate Web Pages: A Large-scale Evaluation of Algorithms. In: The 29th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2006), pp. 284\u2013291 (2006)","DOI":"10.1145\/1148170.1148222"},{"issue":"3","key":"18_CR13","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1002\/asi.10170","volume":"54","author":"T.C. Hoad","year":"2003","unstructured":"Hoad, T.C., Zobel, J.: Methods for Identifying Versioned and Plagiarized Documents. Journal of the American Society for Information Science and Technology\u00a054(3), 203\u2013215 (2003)","journal-title":"Journal of the American Society for Information Science and Technology"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Huffman, S.B., Lehman, A.R., Stolboushkin, A.P., Wong-Toi, H., Yang, F., Roehrig, H.: Multiple-signal Duplicate Detection for Search Evaluation. In: The 30th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2007), pp. 223\u2013230 (2007)","DOI":"10.1145\/1277741.1277782"},{"key":"18_CR15","unstructured":"Manber, U.: Finding Similar Files in a Large File System. In: USENIX Winter Technical Conference, pp. 1\u201310 (1994)"},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Manku, G.S., Jain, A., Sarma, A.D.: Detecting Near-duplicates for Web Crawling. In: The 16th International Conference on World Wide Web (WWW 2007), pp. 141\u2013150 (2007)","DOI":"10.1145\/1242572.1242592"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"Metzler, D., Bernstein, Y., Croft, W.B., Moffat, A., Zobel, J.: Similarity Measures for Tracking Information Flow. In: The 14th ACM Conference on Information and Knowledge Management (CIKM 2005), pp. 517\u2013524 (2005)","DOI":"10.1145\/1099554.1099695"},{"key":"18_CR18","unstructured":"NIST. Secure hash standard. Federal Information Processing Standards, FIPS 180-1 (1995)"},{"key":"18_CR19","unstructured":"NTCIR (NII Test Collection for IR Systems) project, http:\/\/research.nii.ac.jp\/ntcir\/ (accessed on January 23, 2009)"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Seo, J., Croft, W.B.: Local Text Reuse Detection. In: The 31st Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2008), pp. 571\u2013578 (2008)","DOI":"10.1145\/1390334.1390432"},{"key":"18_CR21","unstructured":"Shivakumar, N., Garcia-Molina, H.: SCAM: A Copy Detection Mechanism for Digital Documents. In: International Conference on Theory and Practice of Digital Libraries (1995)"},{"key":"18_CR22","doi-asserted-by":"crossref","unstructured":"Theobald, M., Siddharth, J., Paepcke, A.: SpotSigs: Robust and Efficient Near Duplicate Detection in Large Web Collections. In: The 31st Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2008), pp. 563\u2013570 (2008)","DOI":"10.1145\/1390334.1390431"},{"key":"18_CR23","doi-asserted-by":"crossref","unstructured":"Xiao, C., Wang, W., Lin, X., Yu, J.X.: Efficient Similarity Joins for Near Duplicate Detection. In: The 17th International Conference on World Wide Web (WWW 2008), pp. 131\u2013140 (2008)","DOI":"10.1145\/1367497.1367516"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Yang, H., Callan, J.: Near-duplicate Detection by Instance-level Constrained Clustering. In: The 29th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR 2006), pp. 421\u2013428 (2006)","DOI":"10.1145\/1148170.1148243"}],"container-title":["Lecture Notes in Computer Science","Information Retrieval Technology"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-04769-5_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,22]],"date-time":"2019-05-22T18:00:58Z","timestamp":1558548058000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-04769-5_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2009]]},"ISBN":["9783642047688","9783642047695"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-04769-5_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2009]]}}}