{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:01:33Z","timestamp":1750309293158,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,6,9]],"date-time":"2024-06-09T00:00:00Z","timestamp":1717891200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,6,9]]},"DOI":"10.1145\/3650203.3663327","type":"proceedings-article","created":{"date-parts":[[2024,5,29]],"date-time":"2024-05-29T20:13:23Z","timestamp":1717013603000},"page":"7-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Interactively Improving ML Data Preparation Code via \"Shadow Pipelines\""],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9884-9517","authenticated-orcid":false,"given":"Stefan","family":"Grafberger","sequence":"first","affiliation":[{"name":"AIRLab, University of Amsterdam, The Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0183-6910","authenticated-orcid":false,"given":"Paul","family":"Groth","sequence":"additional","affiliation":[{"name":"University of Amsterdam, The Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4722-5840","authenticated-orcid":false,"given":"Sebastian","family":"Schelter","sequence":"additional","affiliation":[{"name":"BIFOLD &amp; TU Berlin, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,6,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Fairlearn: A toolkit for assessing and improving fairness in AI. Microsoft, Tech. Rep. MSR-TR-2020-32","author":"Bird Sarah","year":"2020","unstructured":"Sarah Bird, Miro Dud\u00edk, Richard Edgar, Brandon Horn, Roman Lutz, Vanessa Milan, Mehrnoosh Sameki, Hanna Wallach, and Kathleen Walker. 2020. Fairlearn: A toolkit for assessing and improving fairness in AI. Microsoft, Tech. Rep. MSR-TR-2020-32 (2020)."},{"key":"e_1_3_2_1_2_1","volume-title":"Data Validation for Machine Learning. MLSys","author":"Breck Eric","year":"2019","unstructured":"Eric Breck, Neoklis Polyzotis, Sudip Roy, Steven Whang, and Martin Zinkevich. 2019. Data Validation for Machine Learning. MLSys (2019)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_4_1","DOI":"10.1109\/ICDE.2019.00139"},{"unstructured":"GitHub. 2021. GitHub Copilot Your AI pair programmer. https:\/\/copilot.github.com\/.","key":"e_1_3_2_1_5_1"},{"key":"e_1_3_2_1_6_1","volume-title":"Automating and Optimizing Data-Centric What-If Analyses on Native Machine Learning Pipelines. SIGMOD","author":"Grafberger Stefan","year":"2023","unstructured":"Stefan Grafberger, Paul Groth, and Sebastian Schelter. 2023. Automating and Optimizing Data-Centric What-If Analyses on Native Machine Learning Pipelines. SIGMOD (2023)."},{"key":"e_1_3_2_1_7_1","volume-title":"Data distribution debugging in machine learning pipelines. VLDBJ","author":"Grafberger Stefan","year":"2022","unstructured":"Stefan Grafberger, Paul Groth, Julia Stoyanovich, and Sebastian Schelter. 2022. Data distribution debugging in machine learning pipelines. VLDBJ (2022)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_8_1","DOI":"10.14778\/3611540.3611606"},{"key":"e_1_3_2_1_9_1","volume-title":"MLINSPECT: A Data Distribution Debugger for Machine Learning Pipelines. SIGMOD","author":"Grafberger Stefan","year":"2021","unstructured":"Stefan Grafberger, Shubha Guha, Julia Stoyanovich, and Sebastian Schelter. 2021. MLINSPECT: A Data Distribution Debugger for Machine Learning Pipelines. SIGMOD (2021)."},{"unstructured":"Grammarly. [n. d.]. Demo. https:\/\/demo.grammarly.com\/.","key":"e_1_3_2_1_10_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_11_1","DOI":"10.1145\/3290605.3300830"},{"unstructured":"Jetbrains. [n.d.]. Code inspections. https:\/\/www.jetbrains.com\/help\/idea\/code-inspection.html#access-inspections-and- settings.","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","volume-title":"Nezihe Merve Gurel, Bo Li, Ce Zhang, Costas J Spanos, and Dawn Song.","author":"Jia Ruoxi","year":"2019","unstructured":"Ruoxi Jia, David Dao, Boxin Wang, Frances Ann Hubis, Nezihe Merve Gurel, Bo Li, Ce Zhang, Costas J Spanos, and Dawn Song. 2019. Efficient task-specific data valuation for nearest neighbor algorithms. PVLDB (2019)."},{"key":"e_1_3_2_1_14_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Karla\u0161 Bojan","year":"2023","unstructured":"Bojan Karla\u0161, David Dao, Matteo Interlandi, Sebastian Schelter, Wentao Wu, and Ce Zhang. 2023. Data Debugging with Shapley Importance over Machine Learning Pipelines. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_15_1","volume-title":"Cleanml: A benchmark for joint data cleaning and machine learning. ICDE","author":"Li Peng","year":"2019","unstructured":"Peng Li, Xi Rao, Jennifer Blase, Yue Zhang, Xu Chu, and Ce Zhang. 2019. Cleanml: A benchmark for joint data cleaning and machine learning. ICDE (2019)."},{"key":"e_1_3_2_1_16_1","volume-title":"Rebecca Isaacs, and Michael Isard.","author":"McSherry Frank","year":"2013","unstructured":"Frank McSherry, Derek Gordon Murray, Rebecca Isaacs, and Michael Isard. 2013. Differential Dataflow.. In CIDR."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_17_1","DOI":"10.18653\/v1"},{"key":"e_1_3_2_1_18_1","volume-title":"Steven Euijong Whang, and Martin Zinkevich","author":"Polyzotis Neoklis","year":"2018","unstructured":"Neoklis Polyzotis, Sudip Roy, Steven Euijong Whang, and Martin Zinkevich. 2018. Data lifecycle challenges in production machine learning: a survey. SIGMOD Record 47, 2 (2018)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_19_1","DOI":"10.1145\/3514221.3517886"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_20_1","DOI":"10.1145\/3448016.3457323"},{"key":"e_1_3_2_1_21_1","volume-title":"On challenges in machine learning model management","author":"Schelter Sebastian","year":"2018","unstructured":"Sebastian Schelter, Felix Biessmann, Tim Januschowski, David Salinas, Stephan Seufert, and Gyuri Szarvas. 2018. On challenges in machine learning model management. IEEE Data Engineering Bulletin (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proactively Screening Machine Learning Pipelines with ArgusEyes. SIGMOD","author":"Schelter Sebastian","year":"2023","unstructured":"Sebastian Schelter, Stefan Grafberger, Shubha Guha, Bojan Karla\u0161, and Ce Zhang. 2023. Proactively Screening Machine Learning Pipelines with ArgusEyes. SIGMOD (2023)."},{"key":"e_1_3_2_1_23_1","volume-title":"Screening Native ML Pipelines with \"ArgusEyes\". CIDR","author":"Schelter Sebastian","year":"2022","unstructured":"Sebastian Schelter, Stefan Grafberger, Shubha Guha, Olivier Sprangers, Bojan Karla\u0161, and Ce Zhang. 2022. Screening Native ML Pipelines with \"ArgusEyes\". CIDR (2022)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.14778\/3229863.3229867"},{"key":"e_1_3_2_1_25_1","volume-title":"JENGA - A Framework to Study the Impact of Data Errors on the Predictions of Machine Learning Models. EDBT","author":"Schelter Sebastian","year":"2021","unstructured":"Sebastian Schelter, Tammo Rukat, and Felix Biessmann. 2021. JENGA - A Framework to Study the Impact of Data Errors on the Predictions of Machine Learning Models. EDBT (2021)."},{"doi-asserted-by":"crossref","unstructured":"Julia Stoyanovich Bill Howe Serge Abiteboul H.V. Jagadish and Sebastian Schelter. 2022. Responsible Data Management. In Communications of the ACM.","key":"e_1_3_2_1_26_1","DOI":"10.1145\/3488717"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_27_1","DOI":"10.1109\/EuroSP.2017.29"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_28_1","DOI":"10.1145\/3318464.3389696"}],"event":{"sponsor":["SIGMOD ACM Special Interest Group on Management of Data"],"acronym":"SIGMOD\/PODS '24","name":"SIGMOD\/PODS '24: International Conference on Management of Data","location":"Santiago AA Chile"},"container-title":["Proceedings of the Eighth Workshop on Data Management for End-to-End Machine Learning"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650203.3663327","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3650203.3663327","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:03:31Z","timestamp":1750291411000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650203.3663327"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,9]]},"references-count":27,"alternative-id":["10.1145\/3650203.3663327","10.1145\/3650203"],"URL":"https:\/\/doi.org\/10.1145\/3650203.3663327","relation":{},"subject":[],"published":{"date-parts":[[2024,6,9]]},"assertion":[{"value":"2024-06-09","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}