{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T10:18:51Z","timestamp":1740133131355,"version":"3.37.3"},"reference-count":53,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2021,4,1]],"date-time":"2021-04-01T00:00:00Z","timestamp":1617235200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,4,1]],"date-time":"2021-04-01T00:00:00Z","timestamp":1617235200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,4,1]],"date-time":"2021-04-01T00:00:00Z","timestamp":1617235200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Parallel Distrib. Syst."],"published-print":{"date-parts":[[2021,4,1]]},"DOI":"10.1109\/tpds.2020.3035170","type":"journal-article","created":{"date-parts":[[2020,11,2]],"date-time":"2020-11-02T20:51:52Z","timestamp":1604350312000},"page":"842-854","source":"Crossref","is-referenced-by-count":1,"title":["SEIZE: Runtime Inspection for Parallel Dataflow Systems"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5806-5120","authenticated-orcid":false,"given":"Youfu","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5756-8321","authenticated-orcid":false,"given":"Matteo","family":"Interlandi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4110-5813","authenticated-orcid":false,"given":"Fotis","family":"Psallidas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carlo","family":"Zaniolo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/237814.237823"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/2882903.2882940"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/568271.223801"},{"key":"ref32","first-page":"448","article-title":"Estimation of query-result distribution and its application in parallel-join load balancing","author":"poosala","year":"1996","journal-title":"Proc Int Conf Very Large Data Bases"},{"article-title":"Practical skew handling in parallel joins","year":"1992","author":"dewitt","key":"ref31"},{"key":"ref30","first-page":"537","article-title":"A taxonomy and performance model of data skew effects in parallel joins","author":"walton","year":"1991","journal-title":"Proc Int Conf Very Large Data Bases"},{"key":"ref37","first-page":"398","article-title":"Efficient computation of frequent and top-k elements in data streams","author":"metwally","year":"2005","journal-title":"Proc 10th Int Conf Database Theory"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2016.7498324"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/2213836.2213840"},{"key":"ref34","first-page":"574","article-title":"Handling data skew in MapReduce","volume":"11","author":"gufler","year":"2011","journal-title":"CLOSER"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/2588555.2594532"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/1242524.1242526"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.14778\/1687553.1687565"},{"key":"ref2","article-title":"Spark: Cluster computing with working sets","author":"zaharia","year":"2010","journal-title":"Proc 2nd USENIX Conf Hot Topics Cloud Comput"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/1327452.1327492"},{"article-title":"Apache spark@ scale: A 60 TB+ production use case","year":"0","author":"kedia","key":"ref20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.14778\/3329772.3329777"},{"article-title":"Alluxio: A virtual distributed file system","year":"2018","author":"li","key":"ref21"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/2463676.2465282"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3389723"},{"key":"ref26","first-page":"525","article-title":"Handling data skew in multiprocessor database computers using partition tuning","author":"hua","year":"1991","journal-title":"Proc Int Conf Very Large Data Bases"},{"key":"ref25","first-page":"49","article-title":"Making state explicit for imperative big data processing","author":"fernandez","year":"2014","journal-title":"Proc USENIX Annu Tech Conf"},{"key":"ref50","first-page":"1","article-title":"Every row counts: Combining sketches and sampling for accurate group-by result estimates","volume":"1","author":"freitag","year":"2019","journal-title":"Rational"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1145\/2517349.2522738"},{"year":"2020","key":"ref53"},{"key":"ref52","first-page":"95","article-title":"SnailTrail: Generalizing critical paths for online analysis of distributed dataflows","author":"hoffmann","year":"2018","journal-title":"Proc 5th USENIX Symp Netw Syst Des Implementation"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/2950290.2983930"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/2884781.2884813"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.2001.1813"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/2465351.2465354"},{"key":"ref13","first-page":"21","article-title":"Re-optimizing data-parallel computing","author":"agarwal","year":"2012","journal-title":"Proc 9th USENIX Conf Netw Syst Des Implementation"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.14778\/2536222.2536223"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/2588555.2610531"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2016.7498324"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.14778\/3231751.3231765"},{"key":"ref18","article-title":"Resilient distributed datasets: A fault-tolerant abstraction for in-memory cluster computing","author":"zaharia","year":"2012","journal-title":"Proc 9th USENIX Conf Netw Syst Des Implementation"},{"key":"ref19","article-title":"Apache spark the fastest open source engine for sorting a petabyte","author":"xin","year":"2014","journal-title":"Databricks Engineering Blog"},{"key":"ref4","first-page":"28","article-title":"Apache flink&#x2122;: Stream and batch processing in a single engine","volume":"38","author":"carbone","year":"2015","journal-title":"IEEE Data Eng Bull"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2011.5767921"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.14778\/2850583.2850595"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/2588555.2594530"},{"key":"ref8","first-page":"273","article-title":"Provenance for generalized map and reduce workflows","author":"ikeda","year":"2011"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3267809.3267814"},{"year":"2020","key":"ref49"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/2523616.2523619"},{"key":"ref46","article-title":"Sketch techniques for approximate query processing","author":"cormode","year":"2011","journal-title":"Foundations and Trends in Databases"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/1989323.1989459"},{"key":"ref48","first-page":"513","article-title":"Is big data performance reproducible in modern cloud networks?","author":"uta","year":"2020","journal-title":"Proc 10th USENIX Symp Netw Syst Des Implementation"},{"key":"ref47","first-page":"409","article-title":"Taming performance variability","author":"maricq","year":"2018","journal-title":"Proc 12th USENIX Symp Operating Syst Des Implementation"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1561\/1900000004"},{"key":"ref41","first-page":"13","article-title":"Sketching streams through the net: Distributed approximate query tracking","author":"cormode","year":"2005","journal-title":"Proc Int Conf Very Large Data Bases"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/s10619-010-7067-2"},{"key":"ref43","first-page":"255","article-title":"Dynamic statistics collection in the teradata unified data architecture","author":"kim","year":"2017","journal-title":"Proc IEEE 33rd Int Conf Data Eng"}],"container-title":["IEEE Transactions on Parallel and Distributed Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/71\/9257114\/09246701.pdf?arnumber=9246701","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T14:50:29Z","timestamp":1652194229000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9246701\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,4,1]]},"references-count":53,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tpds.2020.3035170","relation":{},"ISSN":["1045-9219","1558-2183","2161-9883"],"issn-type":[{"type":"print","value":"1045-9219"},{"type":"electronic","value":"1558-2183"},{"type":"electronic","value":"2161-9883"}],"subject":[],"published":{"date-parts":[[2021,4,1]]}}}