{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T10:11:13Z","timestamp":1781086273756,"version":"3.54.1"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,6,4]],"date-time":"2019-06-04T00:00:00Z","timestamp":1559606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2019,6,4]],"date-time":"2019-06-04T00:00:00Z","timestamp":1559606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61473194"],"award-info":[{"award-number":["61473194"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key R&D Program of China","award":["2017YFC0822604-2"],"award-info":[{"award-number":["2017YFC0822604-2"]}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2016T90799"],"award-info":[{"award-number":["2016T90799"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Scientific Research Foundation of Shenzhen University for Newly-introduced Teachers","award":["2018060"],"award-info":[{"award-number":["2018060"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Big Data"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1186\/s40537-019-0205-4","type":"journal-article","created":{"date-parts":[[2019,6,4]],"date-time":"2019-06-04T09:10:14Z","timestamp":1559639414000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":22,"title":["Exploring and cleaning big data with random sample data blocks"],"prefix":"10.1186","volume":"6","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6750-003X","authenticated-orcid":false,"given":"Salman","family":"Salloum","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joshua Zhexue","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yulin","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,6,4]]},"reference":[{"issue":"1","key":"205_CR1","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1145\/2688072","volume":"58","author":"R Nair","year":"2014","unstructured":"Nair R. Big data needs approximate computing: technical perspective. Commun ACM. 2014;58(1):104. \n                    https:\/\/doi.org\/10.1145\/2688072\n                    \n                  .","journal-title":"Commun ACM"},{"key":"205_CR2","doi-asserted-by":"crossref","unstructured":"Goiri \u00cd, Bianchini R, Nagarakatte S, Nguyen TD. Approxhadoop: bringing approximations to mapreduce frameworks. In: Proceedings of the twentieth international conference on architectural support for programming languages and operating systems. ACM; 2015. p. 383\u201397.","DOI":"10.1145\/2694344.2694351"},{"key":"205_CR3","doi-asserted-by":"publisher","unstructured":"Salloum S, Huang JZ, He Y. Empirical analysis of asymptotic ensemble learning for big data. In: Proceedings of the 3rd IEEE\/ACM international conference on big data computing, applications and technologies. BDCAT \u201916, ACM, New York, NY, USA; 2016, p. 8\u201317. \n                    https:\/\/doi.org\/10.1145\/3006299.3006306\n                    \n                  .","DOI":"10.1145\/3006299.3006306"},{"issue":"1","key":"205_CR4","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1109\/MCG.2017.6","volume":"37","author":"BC Kwon","year":"2017","unstructured":"Kwon BC, Verma J, Haas PJ, Demiralp. Sampling for scalable visual analytics. IEEE Comput Graph Appl. 2017;37(1):100\u20138. \n                    https:\/\/doi.org\/10.1109\/MCG.2017.6\n                    \n                  .","journal-title":"IEEE Comput Graph Appl"},{"key":"205_CR5","doi-asserted-by":"publisher","first-page":"516","DOI":"10.1007\/978-3-662-44845-8_48","volume-title":"Machine Learning and Knowledge Discovery in Databases","author":"Matteo Riondato","year":"2014","unstructured":"Riondato M. Sampling-based data mining algorithms: modern techniques and case studies. In: Proceedings of the 2014th European conference on machine learning and knowledge discovery in databases-volume part III. ECMLPKDD\u201914, Berlin, Heidelberg: Springer; 2014, p. 516\u20139."},{"issue":"3","key":"205_CR6","first-page":"59","volume":"38","author":"S Krishnan","year":"2015","unstructured":"Krishnan S, Wang J, Franklin MJ, Goldberg K, Kraska T, Milo T, Wu E. Sampleclean: fast and reliable analytics on dirty data. IEEE Data Eng Bull. 2015;38(3):59\u201375.","journal-title":"IEEE Data Eng Bull"},{"key":"205_CR7","doi-asserted-by":"publisher","unstructured":"Sutton CA, Hobson T, Geddes J, Caruana R. Data diff: interpretable, executable summaries of changes in distributions for data wrangling. In: Proceedings of the 24th ACM SIGKDD international conference on knowledge discovery & data mining, KDD 2018, London, UK, August 19\u201323, 2018; 2018. p. 2279\u201388. \n                    https:\/\/doi.org\/10.1145\/3219819.3220057\n                    \n                  .","DOI":"10.1145\/3219819.3220057"},{"key":"205_CR8","doi-asserted-by":"publisher","unstructured":"Chu X, Ilyas IF, Krishnan S, Wang J. Data cleaning: overview and emerging challenges. In: Proceedings of the 2016 international conference on management of data. SIGMOD \u201916. ACM, New York, NY, USA; 2016, p. 2201\u20136. \n                    https:\/\/doi.org\/10.1145\/2882903.2912574\n                    \n                  .","DOI":"10.1145\/2882903.2912574"},{"issue":"12","key":"205_CR9","doi-asserted-by":"publisher","first-page":"948","DOI":"10.14778\/2994509.2994514","volume":"9","author":"S Krishnan","year":"2016","unstructured":"Krishnan S, Wang J, Wu E, Franklin MJ, Goldberg K. Activeclean: interactive data cleaning for statistical modeling. Proc VLDB Endow. 2016;9(12):948\u201359. \n                    https:\/\/doi.org\/10.14778\/2994509.2994514\n                    \n                  .","journal-title":"Proc VLDB Endow"},{"key":"205_CR10","unstructured":"Tukey JW. Exploratory data analysis. Behavioral science: quantitative methods. Addison-Wesley, Reading: Mass; 1977."},{"key":"205_CR11","doi-asserted-by":"publisher","unstructured":"Rojas JAR, Kery MB, Rosenthal S, Dey A. Sampling techniques to improve big data exploration. In: 2017 IEEE 7th symposium on large data analysis and visualization (LDAV); 2017, p. 26\u201335. \n                    https:\/\/doi.org\/10.1109\/LDAV.2017.8231848\n                    \n                  .","DOI":"10.1109\/LDAV.2017.8231848"},{"key":"205_CR12","doi-asserted-by":"publisher","unstructured":"Idreos S, Papaemmanouil O, Chaudhuri S. Overview of data exploration techniques. In: Proceedings of the 2015 ACM SIGMOD international conference on management of data. SIGMOD \u201915. ACM, New York, NY, USA; 2015, p. 277\u201381. \n                    https:\/\/doi.org\/10.1145\/2723372.2731084\n                    \n                  .","DOI":"10.1145\/2723372.2731084"},{"issue":"4","key":"205_CR13","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1177\/1473871611415994","volume":"10","author":"S Kandel","year":"2011","unstructured":"Kandel S, Heer J, Plaisant C, Kennedy J, van Ham F, Riche NH, Weaver C, Lee B, Brodbeck D, Buono P. Research directions in data wrangling: visualizations and transformations for usable and credible data. Inf Vis. 2011;10(4):271\u201388. \n                    https:\/\/doi.org\/10.1177\/1473871611415994\n                    \n                  .","journal-title":"Inf Vis"},{"key":"205_CR14","doi-asserted-by":"publisher","DOI":"10.1002\/0471448354","volume-title":"Exploratory data mining and data cleaning","author":"T Dasu","year":"2003","unstructured":"Dasu T, Johnson T. Exploratory data mining and data cleaning. 1st ed. New York: Wiley; 2003.","edition":"1"},{"issue":"12","key":"205_CR15","doi-asserted-by":"publisher","first-page":"993","DOI":"10.14778\/2994509.2994518","volume":"9","author":"Z Abedjan","year":"2016","unstructured":"Abedjan Z, Chu X, Deng D, Fernandez RC, Ilyas IF, Ouzzani M, Papotti P, Stonebraker M, Tang N. Detecting data errors: where are we and what needs to be done? Proc VLDB Endow. 2016;9(12):993\u20131004. \n                    https:\/\/doi.org\/10.14778\/2994509.2994518\n                    \n                  .","journal-title":"Proc VLDB Endow"},{"key":"205_CR16","unstructured":"Hellerstein JM. Quantitative data cleaning for large databases; 2008."},{"key":"205_CR17","doi-asserted-by":"publisher","unstructured":"Krishnan S, Haas D, Franklin MJ, Wu E. Towards reliable interactive data cleaning: a user survey and recommendations. In: Proceedings of the workshop on human-in-the-loop data analytics. HILDA \u201916. ACM, New York, NY, USA; 2016, p. 9\u2013195. \n                    https:\/\/doi.org\/10.1145\/2939502.2939511\n                    \n                  .","DOI":"10.1145\/2939502.2939511"},{"issue":"2","key":"205_CR18","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1145\/3299887.3299891","volume":"47","author":"N Polyzotis","year":"2018","unstructured":"Polyzotis N, Roy S, Whang SE, Zinkevich M. Data lifecycle challenges in production machine learning: a survey. SIGMOD Rec. 2018;47(2):17\u201328. \n                    https:\/\/doi.org\/10.1145\/3299887.3299891\n                    \n                  .","journal-title":"SIGMOD Rec"},{"issue":"4","key":"205_CR19","doi-asserted-by":"publisher","first-page":"281","DOI":"10.1561\/1900000045","volume":"5","author":"IF Ilyas","year":"2015","unstructured":"Ilyas IF, Chu X. Trends in cleaning relational data: consistency and deduplication. Found Trends Databases. 2015;5(4):281\u2013393. \n                    https:\/\/doi.org\/10.1561\/1900000045\n                    \n                  .","journal-title":"Found Trends Databases"},{"key":"205_CR20","doi-asserted-by":"crossref","unstructured":"Park Y, Cafarella MJ, Mozafari B. Visualization-aware sampling for very large databases; 2015. CoRR \n                    arXiv:abs\/1510.03921\n                    \n                  .","DOI":"10.1109\/ICDE.2016.7498287"},{"key":"205_CR21","doi-asserted-by":"crossref","unstructured":"Fisher D. Big data exploration requires collaboration between visualization and data infrastructures. In: HILDA \u201916 proceedings of the workshop on human-in-the-loop data analytics, San Francisco, California, 26 June\u20131 July 2016. New York: ACM; 2016. \n                    https:\/\/dl.acm.org\/citation.cfm?id=2939518\n                    \n                  .","DOI":"10.1145\/2939502.2939518"},{"issue":"3","key":"205_CR22","doi-asserted-by":"publisher","first-page":"311","DOI":"10.3233\/MGS-170273","volume":"13","author":"Y Wang","year":"2017","unstructured":"Wang Y, Zhong Y, Ma Q, Yang G. Distributed and parallel construction method for equi-width histogram in cloud database. Multiagent Grid Syst. 2017;13(3):311\u201329.","journal-title":"Multiagent Grid Syst"},{"key":"205_CR23","unstructured":"Yang P. Ray dataframes: a library for parallel data analysis. Master\u2019s thesis, EECS Department, University of California, Berkeley; May 2018. \n                    http:\/\/www2.eecs.berkeley.edu\/Pubs\/TechRpts\/2018\/EECS-2018-84.html\n                    \n                  . Accessed 15 Jan 2019."},{"key":"205_CR24","volume-title":"R for data science: import, tidy, transform, visualize, and model data","author":"H Wickham","year":"2017","unstructured":"Wickham H, Grolemund G. R for data science: import, tidy, transform, visualize, and model data. 1st ed. Sebastopol: O\u2019Reilly Media Inc; 2017.","edition":"1"},{"key":"205_CR25","volume-title":"Python for data analysis","author":"M Wes","year":"2012","unstructured":"Wes M. Python for data analysis. 1st ed. Sebastopol: O\u2019Reilly Media Inc; 2012.","edition":"1"},{"issue":"1","key":"205_CR26","doi-asserted-by":"publisher","first-page":"93","DOI":"10.1109\/MDAT.2013.2294466","volume":"31","author":"FT Chong","year":"2014","unstructured":"Chong FT, Heck MJR, Ranganathan P, Saleh AAM, Wassel HMG. Data center energy efficiency: improving energy efficiency in data centers beyond technology scaling. IEEE Design Test. 2014;31(1):93\u2013104. \n                    https:\/\/doi.org\/10.1109\/MDAT.2013.2294466\n                    \n                  .","journal-title":"IEEE Design Test"},{"key":"205_CR27","doi-asserted-by":"publisher","unstructured":"Shvachko K, Kuang H, Radia S, Chansler R. The hadoop distributed file system. In: 2010 IEEE 26th symposium on mass storage systems and technologies (MSST); 2010, p. 1\u201310. \n                    https:\/\/doi.org\/10.1109\/MSST.2010.5496972\n                    \n                  .","DOI":"10.1109\/MSST.2010.5496972"},{"issue":"1","key":"205_CR28","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/1327452.132749","volume":"51","author":"J Dean","year":"2008","unstructured":"Dean J, Ghemawat S. Mapreduce: simplified data processing on large clusters. Commun ACM. 2008;51(1):107\u201313. \n                    https:\/\/doi.org\/10.1145\/1327452.132749\n                    \n                  .","journal-title":"Commun ACM"},{"key":"205_CR29","doi-asserted-by":"publisher","unstructured":"Zaharia M, Chowdhury M, Das T, Dave A. Resilient distributed datasets: a fault-tolerant abstraction for in-memory cluster computing. In: NSDI\u201912 Proceedings of the 9th USENIX conference on networked systems design and implementation; 2012, p. 2. \n                    https:\/\/doi.org\/10.1111\/j.1095-8649.2005.00662.x\n                    \n                  .","DOI":"10.1111\/j.1095-8649.2005.00662.x"},{"issue":"11","key":"205_CR30","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1145\/2934664","volume":"59","author":"M Zaharia","year":"2016","unstructured":"Zaharia M, Xin RS, Wendell P, Das T, Armbrust M, Dave A, Meng X, Rosen J, Venkataraman S, Franklin MJ, Ghodsi A, Gonzalez J, Shenker S, Stoica I. Apache spark: a unified engine for big data processing. Commun ACM. 2016;59(11):56\u201365. \n                    https:\/\/doi.org\/10.1145\/2934664\n                    \n                  .","journal-title":"Commun ACM"},{"issue":"3","key":"205_CR31","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1007\/s41060-016-0027-9","volume":"1","author":"S Salloum","year":"2016","unstructured":"Salloum S, Dautov R, Chen X, Peng PX, Huang JZ. Big data analytics on apache spark. Int J Data Sci Anal. 2016;1(3):145\u201364. \n                    https:\/\/doi.org\/10.1007\/s41060-016-0027-9\n                    \n                  .","journal-title":"Int J Data Sci Anal"},{"key":"205_CR32","unstructured":"Bengfort B, Kim J. Data analytics with Hadoop: an introduction for data scientists. Sebastopol: O\u2019Reilly Media; 2016. \n                    https:\/\/books.google.com.hk\/books?id=ou9FDAAAQBAJ\n                    \n                  ."},{"issue":"2","key":"205_CR33","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1007\/s41019-016-0011-3","volume":"1","author":"S Siuly","year":"2016","unstructured":"Siuly S, Zhang Y. Medical big data: neurological diseases diagnosis through medical data analysis. Data Sci Eng. 2016;1(2):54\u201364. \n                    https:\/\/doi.org\/10.1007\/s41019-016-0011-3\n                    \n                  .","journal-title":"Data Sci Eng"},{"key":"205_CR34","unstructured":"Jacobs B. White paper: accelerating R analytics with Spark and Microsoft R Server for Hadoop. White Paper; 2016. \n                    https:\/\/info.microsoft.com\/rs\/157-GQE-382\/images\/EN-CNTNT-Whitepaper-Spark-Microsoft-R-Server-Hadoop.pdf\n                    \n                  . Accessed 17 May 2019."},{"issue":"4","key":"205_CR35","doi-asserted-by":"publisher","first-page":"328","DOI":"10.1007\/s41019-017-0043-3","volume":"2","author":"G Vargas-Solar","year":"2017","unstructured":"Vargas-Solar G, Zechinelli-Martini JL, Espinosa-Oviedo JA. Big data management: what to keep from the past to face future challenges? Data Sci Eng. 2017;2(4):328\u201345. \n                    https:\/\/doi.org\/10.1007\/s41019-017-0043-3\n                    \n                  .","journal-title":"Data Sci Eng"},{"key":"205_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TBDATA.2017.2723473","volume":"99","author":"S Dolev","year":"2017","unstructured":"Dolev S, Florissi P, Gudes E, Sharma S, Singer I. A survey on geographically distributed big-data processing using mapReduce. IEEE Trans Big Data. 2017;99:1. \n                    https:\/\/doi.org\/10.1109\/TBDATA.2017.2723473\n                    \n                  .","journal-title":"IEEE Trans Big Data"},{"key":"205_CR37","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2019.2912723","author":"S Salloum","year":"2019","unstructured":"Salloum S, Huang JZ, He Y. Random sample partition: a distributed data model for big data analysis. IEEE Trans Ind Inf. 2019. \n                    https:\/\/doi.org\/10.1109\/TII.2019.2912723\n                    \n                  .","journal-title":"IEEE Trans Ind Inf."},{"key":"205_CR38","doi-asserted-by":"crossref","unstructured":"Ci X, Meng X. In: Dong XL, Yu X, Li J, Sun Y (eds) An efficient block sampling strategy for online aggregation in the cloud. Cham: Springer; 2015. p. 362\u201373.","DOI":"10.1007\/978-3-319-21042-1_29"},{"key":"205_CR39","doi-asserted-by":"publisher","unstructured":"Chaudhuri S, Das G, Srivastava U. Effective use of block-level sampling in statistics estimation. In: Proceedings of the 2004 ACM SIGMOD international conference on management of data. SIGMOD \u201904. ACM, New York, NY, USA; 2004, p. 287\u201398. \n                    https:\/\/doi.org\/10.1145\/1007568.1007602\n                    \n                  .","DOI":"10.1145\/1007568.1007602"},{"key":"205_CR40","doi-asserted-by":"publisher","unstructured":"Kalavri V, Brundza V, Vlassov V. Block sampling: efficient accurate online aggregation in mapreduce. In: 2013 IEEE 5th international conference on cloud computing technology and science, vol. 1; 2013, p. 250\u201357. \n                    https:\/\/doi.org\/10.1109\/CloudCom.2013.40\n                    \n                  .","DOI":"10.1109\/CloudCom.2013.40"},{"issue":"3","key":"205_CR41","doi-asserted-by":"publisher","first-page":"311","DOI":"10.3233\/MGS-170273","volume":"13","author":"Y Wang","year":"2017","unstructured":"Wang Y, Zhong Y, Ma Q, Yang G. Distributed and parallel construction method for equi-width histogram in cloud database. Multiagent Grid Syst. 2017;13(3):311\u201329. \n                    https:\/\/doi.org\/10.3233\/MGS-170273\n                    \n                  .","journal-title":"Multiagent Grid Syst"},{"key":"205_CR42","doi-asserted-by":"publisher","first-page":"347","DOI":"10.1007\/978-3-319-94295-7_24","volume-title":"Cloud Computing\u2014CLOUD 2018","author":"C Wei","year":"2018","unstructured":"Wei C, Salloum S, Emara TZ, Zhang X, Huang JZ, He Y. A two-stage data processing algorithm to generate random sample partitions for big data analysis. In: Luo M, Zhang L-J, editors. Cloud Computing\u2014CLOUD 2018. Cham: Springer; 2018. p. 347\u201364."},{"key":"205_CR43","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1016\/j.jss.2018.11.007","volume":"148","author":"TZ Emara","year":"2019","unstructured":"Emara TZ, Huang JZ. A distributed data management system to support large-scale data analysis. J Syst Softw. 2019;148:105\u201315. \n                    https:\/\/doi.org\/10.1016\/j.jss.2018.11.007\n                    \n                  .","journal-title":"J Syst Softw"},{"key":"205_CR44","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2889355","author":"S Salloum","year":"2018","unstructured":"Salloum S, Huang JZ, He Y, Chen X. An asymptotic ensemble learning framework for big data analysis. IEEE Access. 2018;. \n                    https:\/\/doi.org\/10.1109\/ACCESS.2018.2889355\n                    \n                  .","journal-title":"IEEE Access"},{"key":"205_CR45","doi-asserted-by":"crossref","unstructured":"Agarwal S, Mozafari B, Panda A, Milner H, Madden S, Stoica, I. Blinkdb: queries with bounded errors and bounded response times on very large data. In: Proceedings of the 8th ACM European Conference on Computer Systems. ACM; 2013, p. 29\u201342.","DOI":"10.1145\/2465351.2465355"},{"key":"205_CR46","doi-asserted-by":"publisher","unstructured":"Krishnan DR, Quoc DL, Bhatotia P, Fetzer C, Rodrigues R. Incapprox: A data analytics system for incremental approximate computing. In: Proceedings of the 25th International Conference on World Wide Web. WWW \u201916. International World Wide Web Conferences Steering Committee, Republic and Canton of Geneva, Switzerland; 2016, p. 1133\u201344. \n                    https:\/\/doi.org\/10.1145\/2872427.2883026\n                    \n                  .","DOI":"10.1145\/2872427.2883026"},{"key":"205_CR47","doi-asserted-by":"publisher","unstructured":"Huang B, Babu S, Yang J. Cumulon: optimizing statistical data analysis in the cloud. In: Proceedings of the 2013 ACM SIGMOD international conference on management of data. SIGMOD \u201913. ACM, New York, NY, USA; 2013, p. 1\u201312. \n                    https:\/\/doi.org\/10.1145\/2463676.2465273\n                    \n                  .","DOI":"10.1145\/2463676.2465273"},{"key":"205_CR48","doi-asserted-by":"publisher","unstructured":"Budiu M, Isaacs R, Murray D, Plotkin G, Barham P, Al-Kiswany S, Boshmaf Y, Luo Q, Andoni A. Interacting with large distributed datasets using sketch. In: Proceedings of the 16th eurographics symposium on parallel graphics and visualization. EGPGV \u201916. Eurographics Association, Goslar Germany, Germany; 2016, p. 31\u201343. \n                    https:\/\/doi.org\/10.2312\/pgv.20161180\n                    \n                  .","DOI":"10.2312\/pgv.20161180"},{"key":"205_CR49","doi-asserted-by":"publisher","unstructured":"Wasay A, Wei X, Dayan N, Idreos S. Data canopy: accelerating exploratory statistical analysis. In: Proceedings of the 2017 ACM international conference on management of data. SIGMOD \u201917. ACM, New York, NY, USA; 2017, p. 557\u201372. \n                    https:\/\/doi.org\/10.1145\/3035918.3064051\n                    \n                  .","DOI":"10.1145\/3035918.3064051"},{"issue":"1","key":"205_CR50","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40537-015-0032-1","volume":"2","author":"S Landset","year":"2015","unstructured":"Landset S, Khoshgoftaar TM, Richter AN, Hasanin T. A survey of open source tools for machine learning with big data in the hadoop ecosystem. J Big Data. 2015;2(1):1\u201336. \n                    https:\/\/doi.org\/10.1186\/s40537-015-0032-1\n                    \n                  .","journal-title":"J Big Data"},{"issue":"1","key":"205_CR51","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1002\/sta4.7","volume":"1","author":"S Guha","year":"2012","unstructured":"Guha S, Hafen R, Rounds J, Xia J, Li J, Xi B, Cleveland WS. Large complex data: divide and recombine (d&r) with rhipe. Stat. 2012;1(1):53\u201367. \n                    https:\/\/doi.org\/10.1002\/sta4.7\n                    \n                  .","journal-title":"Stat"},{"issue":"4","key":"205_CR52","doi-asserted-by":"publisher","first-page":"795","DOI":"10.1111\/rssb.12050","volume":"76","author":"A Kleiner","year":"2014","unstructured":"Kleiner A, Talwalkar A, Sarkar P, Jordan MI. A scalable bootstrap for massive data. J R Stat Soc Series B Stat Methodol. 2014;76(4):795\u2013816. \n                    https:\/\/doi.org\/10.1111\/rssb.12050\n                    \n                  .","journal-title":"J R Stat Soc Series B Stat Methodol"},{"issue":"4","key":"205_CR53","doi-asserted-by":"publisher","first-page":"764","DOI":"10.1016\/j.jesp.2013.03.013","volume":"49","author":"C Leys","year":"2013","unstructured":"Leys C, Ley C, Klein O, Bernard P, Licata L. Detecting outliers: do not use standard deviation around the mean, use absolute deviation around the median. J Exp Soc Psychol. 2013;49(4):764\u20136. \n                    https:\/\/doi.org\/10.1016\/j.jesp.2013.03.013\n                    \n                  .","journal-title":"J Exp Soc Psychol"},{"issue":"4","key":"205_CR54","doi-asserted-by":"publisher","first-page":"300","DOI":"10.14778\/2856318.2856325","volume":"9","author":"N Prokoshyna","year":"2015","unstructured":"Prokoshyna N, Szlichta J, Chiang F, Miller RJ, Srivastava D. Combining quantitative and logical data cleaning. Proc VLDB Endow. 2015;9(4):300\u201311. \n                    https:\/\/doi.org\/10.14778\/2856318.2856325\n                    \n                  .","journal-title":"Proc VLDB Endow"},{"key":"205_CR55","unstructured":"Rezig EK, Ouzzani M, Elmagarmid AK, Aref WG. Human-centric data cleaning [vision]. 2017. CoRR \n                    arXiv:abs\/1712.08971\n                    \n                  ."},{"key":"205_CR56","doi-asserted-by":"publisher","unstructured":"Doan, A. Human-in-the-loop data analysis: A personal perspective. In: Proceedings of the workshop on human-in-the-loop data analytics. HILDA\u201918. ACM, New York, NY, USA; 2018, p. 1\u2013116. \n                    https:\/\/doi.org\/10.1145\/3209900.3209913\n                    \n                  .","DOI":"10.1145\/3209900.3209913"},{"key":"205_CR57","doi-asserted-by":"publisher","unstructured":"Liu J, Wilson A, Gunning D. Workflow-based human-in-the-loop data analytics. In: Proceedings of the 2014 workshop on human centered big data research. HCBDR \u201914. ACM, New York, NY, USA; 2014, p. 49\u2013494952. \n                    https:\/\/doi.org\/10.1145\/2609876.2609888\n                    \n                  .","DOI":"10.1145\/2609876.2609888"},{"issue":"4","key":"205_CR58","first-page":"62","volume":"39","author":"MR Anderson","year":"2016","unstructured":"Anderson MR, Antenucci D, Cafarella MJ. Runtime support for human-in-the-loop feature engineering system. IEEE Data Eng Bull. 2016;39(4):62\u201384.","journal-title":"IEEE Data Eng Bull"},{"issue":"9","key":"205_CR59","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1145\/3080008","volume":"60","author":"G Cormode","year":"2017","unstructured":"Cormode G. Data sketching. Commun ACM. 2017;60(9):48\u201355. \n                    https:\/\/doi.org\/10.1145\/3080008\n                    \n                  .","journal-title":"Commun ACM"}],"container-title":["Journal of Big Data"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s40537-019-0205-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1186\/s40537-019-0205-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s40537-019-0205-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,6,2]],"date-time":"2020-06-02T23:07:52Z","timestamp":1591139272000},"score":1,"resource":{"primary":{"URL":"https:\/\/journalofbigdata.springeropen.com\/articles\/10.1186\/s40537-019-0205-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,6,4]]},"references-count":59,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["205"],"URL":"https:\/\/doi.org\/10.1186\/s40537-019-0205-4","relation":{},"ISSN":["2196-1115"],"issn-type":[{"value":"2196-1115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,6,4]]},"assertion":[{"value":"21 February 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 May 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 June 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"45"}}