{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T06:15:27Z","timestamp":1769926527062,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":70,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,6,10]],"date-time":"2022-06-10T00:00:00Z","timestamp":1654819200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1845638, 2008295, 2106197, 2103794"],"award-info":[{"award-number":["1845638, 2008295, 2106197, 2103794"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Google Research Award"},{"name":"Amazon Research Award"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,6,10]]},"DOI":"10.1145\/3514221.3517854","type":"proceedings-article","created":{"date-parts":[[2022,6,12]],"date-time":"2022-06-12T02:33:49Z","timestamp":1655001229000},"page":"399-413","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Reptile: Aggregation-level Explanations for Hierarchical Data"],"prefix":"10.1145","author":[{"given":"Zezhou","family":"Huang","sequence":"first","affiliation":[{"name":"Columbia University, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eugene","family":"Wu","sequence":"additional","affiliation":[{"name":"Columbia University, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,6,11]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Firas Abuzaid Peter Kraft Sahaana Suri Edward Gan Eric Xu Atul Shenoy Asvin Ananthanarayan John Sheu Erik Meijer Xi Wu et al. 2020. DIFF: a relational interface for large-scale data explanation. The VLDB Journal (2020) 1--26.","DOI":"10.1007\/s00778-020-00633-6"},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Extending Database Technology","volume":"2016","author":"Ainy Eleanor","year":"2016","unstructured":"Eleanor Ainy, Pierre Bourhis, Susan B Davidson, Daniel Deutch, and Tova Milo. 2016. PROX: Approximated Summarization of Data Provenance. In Advances in database technology: proceedings. International Conference on Extending Database Technology, Vol. 2016. NIH Public Access, 620."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.2307\/2981882"},{"key":"e_1_3_2_1_4_1","volume-title":"Selected papers of hirotugu akaike","author":"Akaike Hirotogu","unstructured":"Hirotogu Akaike. 1998. Information theory and an extension of the maximum likelihood principle. In Selected papers of hirotugu akaike. Springer, 199--213."},{"key":"e_1_3_2_1_5_1","volume-title":"Summarizing Provenance of Aggregate Query Results in Relational Databases. In 2021 IEEE 37th International Conference on Data Engineering (ICDE). IEEE","author":"AlOmeir Omar","year":"2021","unstructured":"Omar AlOmeir, Eugenie Yujing Lai, Mostafa Milani, and Rachel Pottinger. 2021. Summarizing Provenance of Aggregate Query Results in Relational Databases. In 2021 IEEE 37th International Conference on Data Engineering (ICDE). IEEE, 1955--1960."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"E. Anderson Z. Bai C. Bischof S. Blackford J. Demmel J. Dongarra J. Du Croz A. Greenbaum S. Hammarling A. McKenney and D. Sorensen. 1999. LAPACK Users' Guide third ed.). Society for Industrial and Applied Mathematics Philadelphia PA.","DOI":"10.1137\/1.9780898719604"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3035918.3035928"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2016.07.005"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1186\/1479-5868-9-110"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/b97636"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2588555.2610520"},{"key":"e_1_3_2_1_12_1","volume-title":"Tim Kraska, and David Karger.","author":"Chepurko Nadiia","year":"2020","unstructured":"Nadiia Chepurko, Ryan Marcus, Emanuel Zgraggen, Raul Castro Fernandez, Tim Kraska, and David Karger. 2020. ARDA: automatic relational data augmentation for machine learning. arXiv preprint arXiv:2003.09758 (2020)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/1077501.1077518"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2882903.2912574"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.14778\/2536258.2536262"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.14778\/2824032.2824109"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3167970"},{"key":"e_1_3_2_1_18_1","volume-title":"International Conference on Artificial Intelligence and Statistics. 2742--2752","author":"Curtin Ryan","year":"2020","unstructured":"Ryan Curtin, Benjamin Moseley, Hung Ngo, XuanLong Nguyen, Dan Olteanu, and Maximilian Schleich. 2020. Rk-means: Fast clustering for relational data. In International Conference on Artificial Intelligence and Statistics. 2742--2752."},{"key":"e_1_3_2_1_19_1","volume-title":"Multilevel analysis in public health research. Annual review of public health","author":"Diez-Roux Ana V","year":"2000","unstructured":"Ana V Diez-Roux. 2000. Multilevel analysis in public health research. Annual review of public health, Vol. 21, 1 (2000), 171--192."},{"key":"e_1_3_2_1_20_1","volume-title":"An interactive web-based dashboard to track COVID-19 in real time. The Lancet infectious diseases","author":"Dong Ensheng","year":"2020","unstructured":"Ensheng Dong, Hongru Du, and Lauren Gardner. 2020. An interactive web-based dashboard to track COVID-19 in real time. The Lancet infectious diseases, Vol. 20, 5 (2020), 533--534."},{"key":"e_1_3_2_1_21_1","volume-title":"PODS '05 .","author":"Fagin Ronald","unstructured":"Ronald Fagin, R. Guha, Ravi Kumar, J. Novak, D. Sivakumar, and A. Tomkins. 2005. Multi-structural databases. In PODS '05 ."},{"key":"e_1_3_2_1_22_1","volume-title":"A multilevel model of life satisfaction: Effects of individual characteristics and neighborhood composition. American Sociological Review","author":"Fernandez Roberto M","year":"1981","unstructured":"Roberto M Fernandez and Jane C Kulik. 1981. A multilevel model of life satisfaction: Effects of individual characteristics and neighborhood composition. American Sociological Review (1981), 840--850."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.learninstruc.2007.09.001"},{"key":"e_1_3_2_1_24_1","volume-title":"Data analysis using regression and multilevel\/hierarchical models","author":"Gelman Andrew","unstructured":"Andrew Gelman and Jennifer Hill. 2006. Data analysis using regression and multilevel\/hierarchical models .Cambridge university press."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3389775"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1561\/9781680838817"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/73.1.43"},{"key":"e_1_3_2_1_28_1","volume-title":"Data cube: A relational aggregation operator generalizing group-by, cross-tab, and sub-totals. Data mining and knowledge discovery","author":"Gray Jim","year":"1997","unstructured":"Jim Gray, Surajit Chaudhuri, Adam Bosworth, Andrew Layman, Don Reichart, Murali Venkatrao, Frank Pellow, and Hamid Pirahesh. 1997. Data cube: A relational aggregation operator generalizing group-by, cross-tab, and sub-totals. Data mining and knowledge discovery, Vol. 1, 1 (1997), 29--53."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3299869.3319888"},{"key":"e_1_3_2_1_30_1","volume-title":"United Nations Economic Commission for Europe (UNECE)","volume":"25","author":"Hellerstein Joseph M","year":"2008","unstructured":"Joseph M Hellerstein. 2008. Quantitative data cleaning for large databases. United Nations Economic Commission for Europe (UNECE), Vol. 25 (2008)."},{"key":"e_1_3_2_1_31_1","volume-title":"A survey of outlier detection methodologies. Artificial intelligence review","author":"Hodge Victoria","year":"2004","unstructured":"Victoria Hodge and Jim Austin. 2004. A survey of outlier detection methodologies. Artificial intelligence review, Vol. 22, 2 (2004), 85--126."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.14778\/2824032.2824103"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF02289635"},{"key":"e_1_3_2_1_34_1","unstructured":"Lukas Kobis. 2017. Learning Decision Trees over Factorized Joins."},{"key":"e_1_3_2_1_35_1","volume-title":"Fifth international workshop on intelligent data analysis in medicine and pharmacology","volume":"1","author":"Laurikkala Jorma","year":"2000","unstructured":"Jorma Laurikkala, Martti Juhola, Erna Kentala, N Lavrac, S Miksch, and B Kavsek. 2000. Informal identification of outliers in medical data. In Fifth international workshop on intelligent data analysis in medicine and pharmacology, Vol. 1. Citeseer, 20--24."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1057\/jibs.2009.30"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1080\/03610910500307695"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/1148928.1700950"},{"key":"e_1_3_2_1_39_1","volume-title":"Isolation Forest. 2008 Eighth IEEE International Conference on Data Mining","author":"Liu F.","year":"2008","unstructured":"F. Liu, K. Ting, and Z. Zhou. 2008. Isolation Forest. 2008 Eighth IEEE International Conference on Data Mining (2008), 413--422."},{"key":"e_1_3_2_1_40_1","volume-title":"Picket: Self-supervised Data Diagnostics for ML Pipelines. arXiv preprint arXiv:2006.04730","author":"Liu Zifan","year":"2020","unstructured":"Zifan Liu, Zhechun Zhou, and Theodoros Rekatsinas. 2020. Picket: Self-supervised Data Diagnostics for ML Pipelines. arXiv preprint arXiv:2006.04730 (2020)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.14778\/3407790.3407801"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3299869.3324956"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1037\/0022-0167.34.4.443"},{"key":"e_1_3_2_1_44_1","unstructured":"Zelda Mariet Rachael Harding Sam Madden et al. 2016. Outlier detection in heterogeneous datasets using automatic tuple expansion. (2016)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/1807167.1807178"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.amepre.2007.09.031"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3299869.3300066"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3183713.3183758"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/345513.345282"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2656335"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.3390\/rs10121887"},{"key":"e_1_3_2_1_52_1","volume-title":"R: A Language and Environment for Statistical Computing","author":"Team R Core","year":"2013","unstructured":"R Core Team. 2013. R: A Language and Environment for Statistical Computing. R Foundation for Statistical Computing, Vienna, Austria. http:\/\/www.R-project.org\/"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.14778\/3137628.3137631"},{"key":"e_1_3_2_1_54_1","volume-title":"Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery","volume":"1","author":"Rousseeuw P.","year":"2011","unstructured":"P. Rousseeuw and M. Hubert. 2011. Robust statistics for outlier detection. Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery, Vol. 1 (2011)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/2588555.2588578"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3183713.3183745"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1037\/0021-9010.90.2.203"},{"key":"e_1_3_2_1_58_1","volume-title":"VLDB","volume":"99","author":"Sarawagi Sunita","year":"1999","unstructured":"Sunita Sarawagi. 1999. Explaining differences in multidimensional aggregates. In VLDB, Vol. 99. Citeseer, 7--10."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1007\/BFb0100984"},{"key":"e_1_3_2_1_60_1","volume-title":"LMFAO: An engine for batches of group-by aggregates. arXiv preprint arXiv:2008.08657","author":"Schleich Maximilian","year":"2020","unstructured":"Maximilian Schleich and Dan Olteanu. 2020. LMFAO: An engine for batches of group-by aggregates. arXiv preprint arXiv:2008.08657 (2020)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/2882903.2882939"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-92bf1922-011"},{"key":"e_1_3_2_1_63_1","volume-title":"Andrew KC Wong, and Mohamed S Kamel","author":"Sun Yanmin","year":"2009","unstructured":"Yanmin Sun, Andrew KC Wong, and Mohamed S Kamel. 2009. Classification of imbalanced data: A review. International journal of pattern recognition and artificial intelligence, Vol. 23, 04 (2009), 687--719."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1093\/ije\/28.5.841"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3035918.3064024"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1177\/0020852314563899"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2014.6816655"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.14778\/2536354.2536356"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/2463676.2463706"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.14778\/1952376.1952378"}],"event":{"name":"SIGMOD\/PODS '22: International Conference on Management of Data","location":"Philadelphia PA USA","acronym":"SIGMOD\/PODS '22","sponsor":["SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 2022 International Conference on Management of Data"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3514221.3517854","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3514221.3517854","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3514221.3517854","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:35Z","timestamp":1750188635000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3514221.3517854"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,10]]},"references-count":70,"alternative-id":["10.1145\/3514221.3517854","10.1145\/3514221"],"URL":"https:\/\/doi.org\/10.1145\/3514221.3517854","relation":{},"subject":[],"published":{"date-parts":[[2022,6,10]]},"assertion":[{"value":"2022-06-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}