{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T12:55:33Z","timestamp":1785416133279,"version":"3.56.0"},"reference-count":14,"publisher":"Association for Computing Machinery (ACM)","issue":"11","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Proc. VLDB Endow."],"published-print":{"date-parts":[[2012,7]]},"abstract":"<jats:p>\n            We introduce the notion of\n            <jats:italic>statistical distortion<\/jats:italic>\n            as an essential metric for measuring the effectiveness of data cleaning strategies. We use this metric to propose a widely applicable yet scalable experimental framework for evaluating data cleaning strategies along three dimensions: glitch improvement, statistical distortion and cost-related criteria. Existing metrics focus on glitch improvement and cost, but not on the statistical impact of data cleaning strategies. We illustrate our framework on real world data, with a comprehensive suite of experiments and analyses.\n          <\/jats:p>","DOI":"10.14778\/2350229.2350279","type":"journal-article","created":{"date-parts":[[2014,6,24]],"date-time":"2014-06-24T12:17:57Z","timestamp":1403612277000},"page":"1674-1683","source":"Crossref","is-referenced-by-count":45,"title":["Statistical distortion"],"prefix":"10.14778","volume":"5","author":[{"given":"Tamraparni","family":"Dasu","sequence":"first","affiliation":[{"name":"AT&amp;T Labs Research, NJ"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ji Meng","family":"Loh","sequence":"additional","affiliation":[{"name":"AT&amp;T Labs Research, NJ"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2012,7]]},"reference":[{"key":"e_1_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020408.2020508"},{"key":"e_1_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1541880.1541883"},{"key":"e_1_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2011.5767864"},{"key":"e_1_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1281100.1281133"},{"key":"e_1_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1314690.1314696"},{"key":"e_1_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1366102.1366103"},{"key":"e_1_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2001.937632"},{"key":"e_1_2_1_8_1","first-page":"932","volume-title":"VLDB","author":"Liu X.","year":"2011"},{"key":"e_1_2_1_9_1","first-page":"1","volume-title":"ICIQ","author":"Loh J.","year":"2011"},{"key":"e_1_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/1807167.1807178"},{"key":"e_1_2_1_11_1","volume-title":"University of California","author":"Olken F.","year":"1993"},{"key":"e_1_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/505248.506010"},{"key":"e_1_2_1_13_1","first-page":"1","volume-title":"CVPR","author":"Shirdhonkar S.","year":"2008"},{"key":"e_1_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/1807167.1807325"}],"container-title":["Proceedings of the VLDB Endowment"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.14778\/2350229.2350279","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,28]],"date-time":"2022-12-28T11:31:50Z","timestamp":1672227110000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.14778\/2350229.2350279"}},"subtitle":["consequences of data cleaning"],"short-title":[],"issued":{"date-parts":[[2012,7]]},"references-count":14,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2012,7]]}},"alternative-id":["10.14778\/2350229.2350279"],"URL":"https:\/\/doi.org\/10.14778\/2350229.2350279","relation":{},"ISSN":["2150-8097"],"issn-type":[{"value":"2150-8097","type":"print"}],"subject":[],"published":{"date-parts":[[2012,7]]}}}