{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T09:58:10Z","timestamp":1756893490746},"reference-count":52,"publisher":"Oxford University Press (OUP)","issue":"11","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["The Computer Journal"],"published-print":{"date-parts":[[2015,11]]},"DOI":"10.1093\/comjnl\/bxv052","type":"journal-article","created":{"date-parts":[[2015,7,19]],"date-time":"2015-07-19T00:09:35Z","timestamp":1437264575000},"page":"3187-3201","source":"Crossref","is-referenced-by-count":17,"title":["Large-Scale Schema-Free Data Deduplication Approach with Adaptive Sliding Window Using MapReduce"],"prefix":"10.1093","volume":"58","author":[{"given":"Kun","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fusen","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"286","published-online":{"date-parts":[[2015,7,18]]},"reference":[{"key":"2015102808332020000_58.11.3187.1","doi-asserted-by":"publisher","DOI":"10.1136\/jech-2012-201767"},{"key":"2015102808332020000_58.11.3187.2","doi-asserted-by":"publisher","DOI":"10.2527\/af.2013-0008"},{"key":"2015102808332020000_58.11.3187.3","doi-asserted-by":"publisher","DOI":"10.1016\/j.datak.2011.07.007"},{"key":"2015102808332020000_58.11.3187.4","doi-asserted-by":"crossref","first-page":"50","DOI":"10.1145\/1721654.1721672","article-title":"A view of cloud computing","volume":"53","author":"Armbrust","year":"2010","journal-title":"Commun. ACM"},{"key":"2015102808332020000_58.11.3187.5","first-page":"17","article-title":"Cloud computing: benefits, risks and recommendations for information security","volume":"72","author":"C","year":"2009","journal-title":"Commun. Comput. Inf. Sci."},{"key":"2015102808332020000_58.11.3187.6","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1155\/2013\/917923","article-title":"A survey on sensor-cloud: architecture, applications, and approaches","volume":"2013","author":"Alamri","year":"2013","journal-title":"Int. J. Distrib. Sens. Netw."},{"key":"2015102808332020000_58.11.3187.7","doi-asserted-by":"crossref","unstructured":"Ma K. , Chen Z. , Abraham A. and Yang B. (2011) A Transparent Data Middleware in Support of Multi-tenancy. Proc. 7th Int. Conf. Next Generation Web Services Practices, pp. 11\u201319.","DOI":"10.1109\/NWeSP.2011.6088144"},{"key":"2015102808332020000_58.11.3187.8","first-page":"25","article-title":"A template-based model transformation approach for deriving multi-tenant SaaS applications","volume":"9","author":"Ma","year":"2012","journal-title":"Acta Polytech. Hung."},{"key":"2015102808332020000_58.11.3187.9","doi-asserted-by":"crossref","first-page":"10 p","DOI":"10.1155\/2014\/583686","article-title":"Multiple wide tables with vertical scalability in multi-tenant sensor cloud systems","volume":"2014","author":"Ma","year":"2014","journal-title":"Int. J. Distrib. Sens. Netw."},{"key":"2015102808332020000_58.11.3187.10","doi-asserted-by":"publisher","DOI":"10.1038\/455028a"},{"key":"2015102808332020000_58.11.3187.11","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1145\/1978915.1978919","article-title":"Scalable SQL and NoSQL data stores","volume":"39","author":"Cattell","year":"2010","journal-title":"ACM SIGMOD Record"},{"key":"2015102808332020000_58.11.3187.12","doi-asserted-by":"publisher","DOI":"10.14778\/2367502.2367564"},{"key":"2015102808332020000_58.11.3187.13","first-page":"72","article-title":"Incremental object matching approach of schema-free data with MapReduce","volume":"36","author":"Ma","year":"2014","journal-title":"Int. J. Comput. Appl."},{"key":"2015102808332020000_58.11.3187.14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-24775-3_75"},{"key":"2015102808332020000_58.11.3187.15","unstructured":"Sik Kim H. and Lee D. (2007) Parallel Linkage. Proc. 16th ACM Conf. Information and Knowledge Management, pp. 283\u2013292."},{"key":"2015102808332020000_58.11.3187.16","doi-asserted-by":"publisher","DOI":"10.14778\/2350229.2350263"},{"key":"2015102808332020000_58.11.3187.17","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2007.250581"},{"key":"2015102808332020000_58.11.3187.18","doi-asserted-by":"publisher","DOI":"10.1016\/j.datak.2009.10.003"},{"key":"2015102808332020000_58.11.3187.19","doi-asserted-by":"publisher","DOI":"10.14778\/1920841.1920904"},{"key":"2015102808332020000_58.11.3187.20","doi-asserted-by":"publisher","DOI":"10.1145\/568271.223807"},{"key":"2015102808332020000_58.11.3187.21","doi-asserted-by":"publisher","DOI":"10.1007\/s00450-011-0177-x"},{"key":"2015102808332020000_58.11.3187.22","doi-asserted-by":"crossref","unstructured":"Zhang Z. , Bhagwat D. , Litwin W. , Long D. and Schwarz S.T. (2012) Improved Deduplication through Parallel Binning. 2012 IEEE 31st Int. Performance Computing and Communications Conf. (IPCCC), pp. 130\u2013141. IEEE.","DOI":"10.1109\/PCCC.2012.6407746"},{"key":"2015102808332020000_58.11.3187.23","doi-asserted-by":"crossref","unstructured":"Xia W. , Jiang H. , Feng D. , Tian L. , Fu M. and Wang Z. (2012) P-dedupe: Exploiting Parallelism in Data Deduplication System. 2012 IEEE 7th Int. Conf. Networking, Architecture and Storage (NAS), pp. 338\u2013347. IEEE.","DOI":"10.1109\/NAS.2012.46"},{"key":"2015102808332020000_58.11.3187.24","doi-asserted-by":"crossref","first-page":"23","DOI":"10.1007\/s13222-012-0110-x","article-title":"Parallel entity resolution with dedoop","volume":"13","author":"Kolb","year":"2013","journal-title":"Datenbank-Spektrum"},{"key":"2015102808332020000_58.11.3187.25","doi-asserted-by":"crossref","first-page":"107","DOI":"10.1145\/1327452.1327492","article-title":"Mapreduce: simplified data processing on large clusters","volume":"51","author":"Cattell","year":"2008","journal-title":"Commun. ACM"},{"key":"2015102808332020000_58.11.3187.26","doi-asserted-by":"crossref","first-page":"72","DOI":"10.1145\/1629175.1629198","article-title":"Mapreduce: a flexible data processing tool","volume":"53","author":"Dean","year":"2010","journal-title":"Commun. ACM"},{"key":"2015102808332020000_58.11.3187.27","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2011.127"},{"key":"2015102808332020000_58.11.3187.28","doi-asserted-by":"crossref","unstructured":"Whang S.E. , Menestrina D. , Koutrika G. , Theobald M. and Garcia-Molina H. (2009) Entity Resolution with Iterative Blocking. Proc. 2009 ACM SIGMOD Int. Conf. Management of Data, pp. 219\u2013232. ACM.","DOI":"10.1145\/1559845.1559870"},{"key":"2015102808332020000_58.11.3187.29","unstructured":"Draisbach U. and Naumann F. (2009) A Comparison and Generalization of Blocking and Windowing Algorithms for Duplicate Detection. Proc. Int. Workshop on Quality in Databases (QDB), pp. 51\u201356."},{"key":"2015102808332020000_58.11.3187.30","unstructured":"Kirsten T. , Kolb L. , Hartung M. , Gross A. , K\u00f6pcke H. and Rahm E. (2010) Data Partitioning for Parallel Entity Matching. 8th Int. Workshop on Quality in Databases."},{"key":"2015102808332020000_58.11.3187.31","first-page":"9","article-title":"Robust record linkage blocking using suffix arrays and bloom filters","volume":"5","author":"De Vries","year":"2011","journal-title":"ACM Trans. Knowl. Discov. Data (TKDD)"},{"key":"2015102808332020000_58.11.3187.32","doi-asserted-by":"crossref","unstructured":"Bilenko M. , Kamath B. and Mooney R.J. (2006) Adaptive Blocking: Learning to Scale Up Record Linkage. 6th Int. Conf. Data Mining, 2006, ICDM'06, pp. 87\u201396. IEEE.","DOI":"10.1109\/ICDM.2006.13"},{"key":"2015102808332020000_58.11.3187.33","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2010.134"},{"key":"2015102808332020000_58.11.3187.34","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2010.234"},{"key":"2015102808332020000_58.11.3187.35","doi-asserted-by":"publisher","DOI":"10.2307\/2286061"},{"key":"2015102808332020000_58.11.3187.36","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1995.476344"},{"key":"2015102808332020000_58.11.3187.37","doi-asserted-by":"crossref","unstructured":"Yan S. , Lee D. , Kan M.-Y. and Giles L.C. (2007) Adaptive Sorted Neighborhood Methods for Efficient Record Linkage. Proc. 7th ACM\/IEEE-CS Joint Conf. Digital libraries, pp. 185\u2013194. ACM.","DOI":"10.1145\/1255175.1255213"},{"key":"2015102808332020000_58.11.3187.38","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/978-3-319-08608-8_1","article-title":"Dynamic Sorted Neighborhood Indexing for Real-time Entity Resolution","volume-title":"Databases Theory Appl.","author":"Ramadan","year":"2014"},{"key":"2015102808332020000_58.11.3187.39","doi-asserted-by":"crossref","unstructured":"Ramadan B. and Christen P. (2014) Forest-based Dynamic Sorted Neighborhood Indexing for Real-time Entity Resolution. Proc. 23rd ACM Int. Conf. Information and Knowledge Management, pp. 1787\u20131790. ACM.","DOI":"10.1145\/2661829.2661869"},{"key":"2015102808332020000_58.11.3187.40","unstructured":"Ma K. and Yang B. (2015) Parallel NoSQL Entity Resolution Approach with Mapreduce. Proc. 7th Int. Conf. Intelligent Networking and Collaborative Systems, pp. 1\u20136."},{"key":"2015102808332020000_58.11.3187.41","doi-asserted-by":"crossref","unstructured":"Cohen W.W. and Richman J. (2002) Learning to Match and Cluster Large High-dimensional Data Sets for Data Integration. Proc. 8th ACM SIGKDD Int. Conf. Knowl. Discovery and Data Mining, pp. 475\u2013480. ACM.","DOI":"10.1145\/775047.775116"},{"key":"2015102808332020000_58.11.3187.42","doi-asserted-by":"crossref","unstructured":"Christen P. (2006) A Comparison of Personal Name Matching: Techniques and Practical Issues. Proc. 6th IEEE Int. Conf. Data Mining Workshops, pp. 290\u2013294. IEEE.","DOI":"10.1109\/ICDMW.2006.2"},{"key":"2015102808332020000_58.11.3187.43","unstructured":"Christen P. et al. (2007) Towards Parameter-free Blocking for Scalable Record Linkage. Department of Computer Science, Faculty of Engineering and Information Technology, Australian National University."},{"key":"2015102808332020000_58.11.3187.44","doi-asserted-by":"crossref","unstructured":"Santos W. , Teixeira T. , Machado C. , Meira W. , Da Silva A.S. , Ferreira D. and Guedes D. (2007) A Scalable Parallel Deduplication Algorithm. 19th Int. Symp. Computer Architecture and High Performance Computing, pp. 79\u201386. IEEE.","DOI":"10.1109\/SBAC-PAD.2007.32"},{"key":"2015102808332020000_58.11.3187.45","doi-asserted-by":"crossref","unstructured":"Dal Bianco G. , Galante R. and Heuser C.A. (2011) A Fast Approach for Parallel Deduplication on Multicore Processors. Proc. 2011 ACM Symp. Applied Computing, pp. 1027\u20131032. ACM.","DOI":"10.1145\/1982185.1982411"},{"key":"2015102808332020000_58.11.3187.46","doi-asserted-by":"crossref","unstructured":"Lin B. , Liao X. , Li S. , Wang Y. , Huang H. and Wen L. (2013) G-paradex: Gpu-based Parallel Indexing for Fast Data Deduplication. Adv. Parallel Processing Technologies, pp. 91\u2013103. Springer.","DOI":"10.1007\/978-3-642-45293-2_7"},{"key":"2015102808332020000_58.11.3187.47","doi-asserted-by":"publisher","DOI":"10.14778\/2367502.2367527"},{"key":"2015102808332020000_58.11.3187.48","doi-asserted-by":"crossref","unstructured":"Gufler B. , Augsten N. , Reiser A. and Kemper A. (2012) Load Balancing in Mapreduce Based on Scalable Cardinality Estimates. 2012 IEEE 28th Int. Conf. Data Engineering (ICDE), pp. 522\u2013533. IEEE.","DOI":"10.1109\/ICDE.2012.58"},{"key":"2015102808332020000_58.11.3187.49","doi-asserted-by":"crossref","unstructured":"Draisbach U. , Naumann F. , Szott S. and Wonneberg O. (2012) Adaptive Windows for Duplicate Detection. Proc. 2012 IEEE 28th Int. Conf. Data Engineering, pp. 1073\u20131083.","DOI":"10.1109\/ICDE.2012.20"},{"key":"2015102808332020000_58.11.3187.50","unstructured":"Dean J. and Ghemawat S. (2004) Mapreduce: Simplified Data Processing on Large Clusters. Proc. 2004 Symp. Operating System Design and Implementation, pp. 1\u201313."},{"key":"2015102808332020000_58.11.3187.51","doi-asserted-by":"crossref","unstructured":"Bhandarkar M. (2010) Mapreduce Programming with Apache Hadoop. Proc. 2010 IEEE Int. Symp. Parallel and Distributed Processing, pp. 1\u20131.","DOI":"10.1109\/IPDPS.2010.5470377"},{"key":"2015102808332020000_58.11.3187.52","doi-asserted-by":"crossref","unstructured":"Vernica R. , Carey M.J. and Li C. (2010) Efficient Parallel Set-similarity Joins Using Mapreduce. Proc. 2010 ACM SIGMOD Int. Conf. Management of Data, pp. 495\u2013506. ACM.","DOI":"10.1145\/1807167.1807222"}],"container-title":["The Computer Journal"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/academic.oup.com\/comjnl\/article-pdf\/58\/11\/3187\/5159799\/bxv052.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,12]],"date-time":"2023-08-12T09:32:59Z","timestamp":1691832779000},"score":1,"resource":{"primary":{"URL":"https:\/\/academic.oup.com\/comjnl\/article-lookup\/doi\/10.1093\/comjnl\/bxv052"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,7,18]]},"references-count":52,"journal-issue":{"issue":"11","published-online":{"date-parts":[[2015,10,28]]},"published-print":{"date-parts":[[2015,11]]}},"alternative-id":["10.1093\/comjnl\/bxv052"],"URL":"https:\/\/doi.org\/10.1093\/comjnl\/bxv052","relation":{},"ISSN":["0010-4620","1460-2067"],"issn-type":[{"value":"0010-4620","type":"print"},{"value":"1460-2067","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,7,18]]}}}