{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,3]],"date-time":"2026-03-03T10:17:17Z","timestamp":1772533037058,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,6,9]],"date-time":"2021-06-09T00:00:00Z","timestamp":1623196800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2040727"],"award-info":[{"award-number":["2040727"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006785","name":"Google","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,6,9]]},"DOI":"10.1145\/3448016.3457274","type":"proceedings-article","created":{"date-parts":[[2021,6,18]],"date-time":"2021-06-18T17:22:39Z","timestamp":1624036959000},"page":"1584-1596","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":17,"title":["Towards Benchmarking Feature Type Inference for AutoML Platforms"],"prefix":"10.1145","author":[{"given":"Vraj","family":"Shah","sequence":"first","affiliation":[{"name":"University of California, San Diego, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jonathan","family":"Lacanlale","sequence":"additional","affiliation":[{"name":"California State University, Northridge, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Premanand","family":"Kumar","sequence":"additional","affiliation":[{"name":"University of California, San Diego, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kevin","family":"Yang","sequence":"additional","affiliation":[{"name":"University of California, San Diego, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arun","family":"Kumar","sequence":"additional","affiliation":[{"name":"University of California, San Diego, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,6,18]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Systems, Challenges","author":"Hutter Frank","year":"2019","unstructured":"Frank Hutter , Lars Kotthoff , and Joaquin Vanschoren , editors. Automated Machine Learning - Methods , Systems, Challenges . The Springer Series on Challenges in Machine Learning. Springer , 2019 . Frank Hutter, Lars Kotthoff, and Joaquin Vanschoren, editors. Automated Machine Learning - Methods, Systems, Challenges. The Springer Series on Challenges in Machine Learning. Springer, 2019."},{"key":"e_1_3_2_2_2_1","volume-title":"Accessed","author":"Google Cloud","year":"2021","unstructured":"Google Cloud AutoML , https:\/\/cloud.google.com\/automl\/ , Accessed March 22, 2021 . Google Cloud AutoML, https:\/\/cloud.google.com\/automl\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_3_1","volume-title":"Accessed","author":"Salesforce Einstein","year":"2021","unstructured":"Salesforce Einstein AutoML , https:\/\/www.salesforce.com\/video\/1776007 , Accessed March 22, 2021 . Salesforce Einstein AutoML, https:\/\/www.salesforce.com\/video\/1776007, Accessed March 22, 2021."},{"key":"e_1_3_2_2_4_1","volume-title":"An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909","author":"Gijsbers Pieter","year":"2019","unstructured":"Pieter Gijsbers , Erin LeDell , Janek Thomas , S\u00e9bastien Poirier , Bernd Bischl , and Joaquin Vanschoren . An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909 , 2019 . Pieter Gijsbers, Erin LeDell, Janek Thomas, S\u00e9bastien Poirier, Bernd Bischl, and Joaquin Vanschoren. An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909, 2019."},{"key":"e_1_3_2_2_5_1","first-page":"1607","volume-title":"Hosagrahar V Jagadish. Foofah: A Programming-By-Example System for Synthesizing Data Transformation Programs. In Proceedings of the 2017 ACM International Conference on Management of Data","author":"Jin Zhongjun","year":"2017","unstructured":"Zhongjun Jin , Michael R Anderson , Michael Cafarella , and Hosagrahar V Jagadish. Foofah: A Programming-By-Example System for Synthesizing Data Transformation Programs. In Proceedings of the 2017 ACM International Conference on Management of Data , pages 1607 -- 1610 , 2017 . Zhongjun Jin, Michael R Anderson, Michael Cafarella, and Hosagrahar V Jagadish. Foofah: A Programming-By-Example System for Synthesizing Data Transformation Programs. In Proceedings of the 2017 ACM International Conference on Management of Data, pages 1607--1610, 2017."},{"key":"e_1_3_2_2_6_1","volume-title":"BoostClean: Automated Error Detection and Repair for Machine Learning. CoRR, abs\/1711.01299","author":"Krishnan Sanjay","year":"2017","unstructured":"Sanjay Krishnan , Michael J. Franklin , Ken Goldberg , and Eugene Wu . BoostClean: Automated Error Detection and Repair for Machine Learning. CoRR, abs\/1711.01299 , 2017 . Sanjay Krishnan, Michael J. Franklin, Ken Goldberg, and Eugene Wu. BoostClean: Automated Error Detection and Repair for Machine Learning. CoRR, abs\/1711.01299, 2017."},{"key":"e_1_3_2_2_7_1","volume-title":"Accessed","year":"2021","unstructured":"TransmogrifAI : Automated Machine Learning for Structured Data, https:\/\/transmogrif.ai\/ , Accessed March 22, 2021 . TransmogrifAI: Automated Machine Learning for Structured Data, https:\/\/transmogrif.ai\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_8_1","unstructured":"Denis Baylor Eric Breck Heng-Tze Cheng Noah Fiedel Chuan Yu Foo Zakaria Haque Salem Haykal Mustafa Ispir Vihan Jain Levent Koc Chiu Yuen Koo Lukasz Lew Clemens Mewald Akshay Naresh Modi Neoklis Polyzotis Sukriti Ramesh Sudip Roy Steven Euijong Whang Martin Wicke Jarek Wilkiewicz Xin Zhang and Martin Zinkevich. TFX: A TensorFlow-Based Production-Scale Machine Learning Platform. In Proceedings of the 23rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining Halifax NS Canada August 13 - 17 2017 pages 1387--1395. ACM 2017.  Denis Baylor Eric Breck Heng-Tze Cheng Noah Fiedel Chuan Yu Foo Zakaria Haque Salem Haykal Mustafa Ispir Vihan Jain Levent Koc Chiu Yuen Koo Lukasz Lew Clemens Mewald Akshay Naresh Modi Neoklis Polyzotis Sukriti Ramesh Sudip Roy Steven Euijong Whang Martin Wicke Jarek Wilkiewicz Xin Zhang and Martin Zinkevich. TFX: A TensorFlow-Based Production-Scale Machine Learning Platform. In Proceedings of the 23rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining Halifax NS Canada August 13 - 17 2017 pages 1387--1395. ACM 2017."},{"key":"e_1_3_2_2_9_1","volume-title":"AutoGluon-Tabular: Robust and Accurate AutoML for Structured Data. CoRR, abs\/2003.06505","author":"Erickson Nick","year":"2020","unstructured":"Nick Erickson , Jonas Mueller , Alexander Shirkov , Hang Zhang , Pedro Larroy , Mu Li , and Alexander J. Smola . AutoGluon-Tabular: Robust and Accurate AutoML for Structured Data. CoRR, abs\/2003.06505 , 2020 . Nick Erickson, Jonas Mueller, Alexander Shirkov, Hang Zhang, Pedro Larroy, Mu Li, and Alexander J. Smola. AutoGluon-Tabular: Robust and Accurate AutoML for Structured Data. CoRR, abs\/2003.06505, 2020."},{"issue":"3","key":"e_1_3_2_2_10_1","first-page":"211","volume":"115","author":"Russakovsky Olga","year":"2015","unstructured":"Olga Russakovsky , Jia Deng , Hao Su , Jonathan Krause , Sanjeev Satheesh , Sean Ma , Zhiheng Huang , Andrej Karpathy , Aditya Khosla , Michael S. Bernstein , Alexander C. Berg , and Fei-Fei Li . ImageNet Large Scale Visual Recognition Challenge. Int. J. Comput. Vis. , 115 ( 3 ): 211 -- 252 , 2015 . Olga Russakovsky, Jia Deng, Hao Su, Jonathan Krause, Sanjeev Satheesh, Sean Ma, Zhiheng Huang, Andrej Karpathy, Aditya Khosla, Michael S. Bernstein, Alexander C. Berg, and Fei-Fei Li. ImageNet Large Scale Visual Recognition Challenge. Int. J. Comput. Vis., 115(3):211--252, 2015.","journal-title":"ImageNet Large Scale Visual Recognition Challenge. Int. J. Comput. Vis."},{"key":"e_1_3_2_2_11_1","first-page":"14","article-title":"a Foundational Python Library for Data Analysis and Statistics","author":"McKinney Wes","year":"2011","unstructured":"Wes McKinney . pandas : a Foundational Python Library for Data Analysis and Statistics . Python for High Performance and Scientific Computing , 14 , 2011 . Wes McKinney. pandas: a Foundational Python Library for Data Analysis and Statistics. Python for High Performance and Scientific Computing, 14, 2011.","journal-title":"Python for High Performance and Scientific Computing"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330993"},{"key":"e_1_3_2_2_13_1","first-page":"35","volume-title":"Yan and Yeye He. Synthesizing Type-Detection Logic for Rich Semantic Data Types Using Open-Source Code. In Proceedings of the 2018 International Conference on Management of Data","author":"Cong","year":"2018","unstructured":"Cong Yan and Yeye He. Synthesizing Type-Detection Logic for Rich Semantic Data Types Using Open-Source Code. In Proceedings of the 2018 International Conference on Management of Data , pages 35 -- 50 , 2018 . Cong Yan and Yeye He. Synthesizing Type-Detection Logic for Rich Semantic Data Types Using Open-Source Code. In Proceedings of the 2018 International Conference on Management of Data, pages 35--50, 2018."},{"key":"e_1_3_2_2_14_1","volume-title":"Accessed","author":"Shah Vraj","year":"2021","unstructured":"Vraj Shah , Kevin Yang , and Arun Kumar . Improving Feature Type Inference Accuracy of TFDV with SortingHat , Accessed March 22, 2021 . https:\/\/adalabucsd.github.io\/papers\/TR_2020_TFDV.pdf. Vraj Shah, Kevin Yang, and Arun Kumar. Improving Feature Type Inference Accuracy of TFDV with SortingHat, Accessed March 22, 2021. https:\/\/adalabucsd.github.io\/papers\/TR_2020_TFDV.pdf."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2641190.2641198"},{"key":"e_1_3_2_2_16_1","volume-title":"Accessed","author":"Standards URL","year":"2021","unstructured":"URL Standards , https:\/\/url.spec.whatwg.org , Accessed March 22, 2021 . URL Standards, https:\/\/url.spec.whatwg.org, Accessed March 22, 2021."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3448016.3457274"},{"key":"e_1_3_2_2_18_1","volume-title":"Overton: A Data System for Monitoring and Improving Machine-Learned Products. arXiv preprint arXiv:1909.05372","author":"R\u00e9 Christopher","year":"2019","unstructured":"Christopher R\u00e9 , Feng Niu , Pallavi Gudipati , and Charles Srisuwananukorn . Overton: A Data System for Monitoring and Improving Machine-Learned Products. arXiv preprint arXiv:1909.05372 , 2019 . Christopher R\u00e9, Feng Niu, Pallavi Gudipati, and Charles Srisuwananukorn. Overton: A Data System for Monitoring and Improving Machine-Learned Products. arXiv preprint arXiv:1909.05372, 2019."},{"key":"e_1_3_2_2_19_1","first-page":"9397","volume-title":"Advances in neural information processing systems","author":"Chen Vincent","year":"2019","unstructured":"Vincent Chen , Sen Wu , Alexander J Ratner , Jen Weng , and Christopher R\u00e9 . Slice-based Learning: A Programming Model for Residual Learning in Critical Data Slices . In Advances in neural information processing systems , pages 9397 -- 9407 , 2019 . Vincent Chen, Sen Wu, Alexander J Ratner, Jen Weng, and Christopher R\u00e9. Slice-based Learning: A Programming Model for Residual Learning in Critical Data Slices. In Advances in neural information processing systems, pages 9397--9407, 2019."},{"key":"e_1_3_2_2_20_1","first-page":"1550","volume-title":"Steven Euijong Whang. Slice Finder: Automated Data Slicing for Model Validation. In 2019 IEEE 35th International Conference on Data Engineering (ICDE)","author":"Chung Yeounoh","year":"2019","unstructured":"Yeounoh Chung , Tim Kraska , Neoklis Polyzotis , Ki Hyun Tae , and Steven Euijong Whang. Slice Finder: Automated Data Slicing for Model Validation. In 2019 IEEE 35th International Conference on Data Engineering (ICDE) , pages 1550 -- 1553 . IEEE, 2019 . Yeounoh Chung, Tim Kraska, Neoklis Polyzotis, Ki Hyun Tae, and Steven Euijong Whang. Slice Finder: Automated Data Slicing for Model Validation. In 2019 IEEE 35th International Conference on Data Engineering (ICDE), pages 1550--1553. IEEE, 2019."},{"key":"e_1_3_2_2_21_1","volume-title":"Slice Tuner: A Selective Data Collection Framework for Accurate and Fair Machine Learning Models. arXiv preprint arXiv:2003.04549","author":"Tae Ki Hyun","year":"2020","unstructured":"Ki Hyun Tae and Steven Euijong Whang . Slice Tuner: A Selective Data Collection Framework for Accurate and Fair Machine Learning Models. arXiv preprint arXiv:2003.04549 , 2020 . Ki Hyun Tae and Steven Euijong Whang. Slice Tuner: A Selective Data Collection Framework for Accurate and Fair Machine Learning Models. arXiv preprint arXiv:2003.04549, 2020."},{"key":"e_1_3_2_2_22_1","volume-title":"Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015","author":"Zhang Xiang","year":"2015","unstructured":"Xiang Zhang , Junbo Jake Zhao , and Yann LeCun . Character-level Convolutional Networks for Text Classification. In Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett, editors , Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015 , December 7 --12 , 2015 , Montreal, Quebec, Canada, pages 649--657, 2015. Xiang Zhang, Junbo Jake Zhao, and Yann LeCun. Character-level Convolutional Networks for Text Classification. In Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett, editors, Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7--12, 2015, Montreal, Quebec, Canada, pages 649--657, 2015."},{"key":"e_1_3_2_2_23_1","volume-title":"Text Understanding from Scratch. CoRR, abs\/1502.01710","author":"Zhang Xiang","year":"2015","unstructured":"Xiang Zhang and Yann LeCun . Text Understanding from Scratch. CoRR, abs\/1502.01710 , 2015 . Xiang Zhang and Yann LeCun. Text Understanding from Scratch. CoRR, abs\/1502.01710, 2015."},{"key":"e_1_3_2_2_24_1","first-page":"1","volume-title":"Prabodh Mishra. The Design and Operation of CloudLab. In Proceedings of the USENIX Annual Technical Conference (ATC)","author":"Duplyakin Dmitry","year":"2019","unstructured":"Dmitry Duplyakin , Robert Ricci , Aleksander Maricq , Gary Wong , Jonathon Duerig , Eric Eide , Leigh Stoller , Mike Hibler , David Johnson , Kirk Webb , Aditya Akella , Kuangching Wang , Glenn Ricart , Larry Landweber , Chip Elliott , Michael Zink , Emmanuel Cecchet , Snigdhaswin Kar , and Prabodh Mishra. The Design and Operation of CloudLab. In Proceedings of the USENIX Annual Technical Conference (ATC) , pages 1 -- 14 , July 2019 . Dmitry Duplyakin, Robert Ricci, Aleksander Maricq, Gary Wong, Jonathon Duerig, Eric Eide, Leigh Stoller, Mike Hibler, David Johnson, Kirk Webb, Aditya Akella, Kuangching Wang, Glenn Ricart, Larry Landweber, Chip Elliott, Michael Zink, Emmanuel Cecchet, Snigdhaswin Kar, and Prabodh Mishra. The Design and Operation of CloudLab. In Proceedings of the USENIX Annual Technical Conference (ATC), pages 1--14, July 2019."},{"key":"e_1_3_2_2_25_1","volume-title":"Accessed","author":"Feature Type Inference Github Repository","year":"2021","unstructured":"Github Repository for ML Feature Type Inference , https:\/\/github.com\/pvn25\/ML-Data-Prep-Zoo\/tree\/master\/MLFeatureTypeInference , Accessed March 22, 2021 . Github Repository for ML Feature Type Inference, https:\/\/github.com\/pvn25\/ML-Data-Prep-Zoo\/tree\/master\/MLFeatureTypeInference, Accessed March 22, 2021."},{"key":"e_1_3_2_2_26_1","volume-title":"Accessed","author":"Tables Google","year":"2021","unstructured":"Google AutoML Tables , https:\/\/cloud.google.com\/automl-tables , Accessed March 22, 2021 . Google AutoML Tables, https:\/\/cloud.google.com\/automl-tables, Accessed March 22, 2021."},{"key":"e_1_3_2_2_27_1","volume-title":"Accessed","year":"2021","unstructured":"DataRobot , https:\/\/www.datarobot.com , Accessed March 22, 2021 . DataRobot, https:\/\/www.datarobot.com, Accessed March 22, 2021."},{"key":"e_1_3_2_2_28_1","volume-title":"Accessed","author":"Trifacta","year":"2021","unstructured":"Trifacta : Data Wrangling Tools & Software, https:\/\/www.trifacta.com\/ , Accessed March 22, 2021 . Trifacta: Data Wrangling Tools & Software, https:\/\/www.trifacta.com\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-21606-5"},{"key":"e_1_3_2_2_30_1","first-page":"133","volume-title":"Proceedings of the first instructional conference on machine learning","volume":"242","author":"Ramos Juan","year":"2003","unstructured":"Juan Ramos . Using TF-IDF to Determine Word Relevance in Document Queries . In Proceedings of the first instructional conference on machine learning , volume 242 , pages 133 -- 142 . New Jersey, USA , 2003 . Juan Ramos. Using TF-IDF to Determine Word Relevance in Document Queries. In Proceedings of the first instructional conference on machine learning, volume 242, pages 133--142. New Jersey, USA, 2003."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.14778\/3157794.3157797"},{"key":"e_1_3_2_2_32_1","first-page":"223","volume-title":"Proceedings of the VLDB Endowment. International Conference on Very Large Data Bases","volume":"12","author":"Varma Paroma","unstructured":"Paroma Varma and Christopher R\u00e9. Snuba : Automating Weak Supervision to Label Training Data . In Proceedings of the VLDB Endowment. International Conference on Very Large Data Bases , volume 12 , page 223 . NIH Public Access, 2018. Paroma Varma and Christopher R\u00e9. Snuba: Automating Weak Supervision to Label Training Data. In Proceedings of the VLDB Endowment. International Conference on Very Large Data Bases, volume 12, page 223. NIH Public Access, 2018."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2487575.2487629"},{"key":"e_1_3_2_2_34_1","volume-title":"Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015","author":"Feurer Matthias","year":"2015","unstructured":"Matthias Feurer , Aaron Klein , Katharina Eggensperger , Jost Tobias Springenberg , Manuel Blum , and Frank Hutter . Efficient and Robust Automated Machine Learning. In Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett, editors , Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015 , December 7 --12 , 2015 , Montreal, Quebec, Canada, pages 2962--2970, 2015. Matthias Feurer, Aaron Klein, Katharina Eggensperger, Jost Tobias Springenberg, Manuel Blum, and Frank Hutter. Efficient and Robust Automated Machine Learning. In Corinna Cortes, Neil D. Lawrence, Daniel D. Lee, Masashi Sugiyama, and Roman Garnett, editors, Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7--12, 2015, Montreal, Quebec, Canada, pages 2962--2970, 2015."},{"key":"e_1_3_2_2_35_1","first-page":"979","volume-title":"Dawn Song. ExploreKit: Automatic Feature Generation and Selection. In 2016 IEEE 16th International Conference on Data Mining (ICDM)","author":"Katz Gilad","year":"2016","unstructured":"Gilad Katz , Eui Chul Richard Shin , and Dawn Song. ExploreKit: Automatic Feature Generation and Selection. In 2016 IEEE 16th International Conference on Data Mining (ICDM) , pages 979 -- 984 . IEEE, 2016 . Gilad Katz, Eui Chul Richard Shin, and Dawn Song. ExploreKit: Automatic Feature Generation and Selection. In 2016 IEEE 16th International Conference on Data Mining (ICDM), pages 979--984. IEEE, 2016."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/DSAA.2015.7344858"},{"key":"e_1_3_2_2_37_1","volume-title":"Accessed","author":"Zipline Airbnb","year":"2021","unstructured":"Airbnb Zipline , https:\/\/conferences.oreilly.com\/strata\/strata-ny-2018\/public\/schedule\/detail\/68114 , Accessed March 22, 2021 . Airbnb Zipline, https:\/\/conferences.oreilly.com\/strata\/strata-ny-2018\/public\/schedule\/detail\/68114, Accessed March 22, 2021."},{"key":"e_1_3_2_2_38_1","volume-title":"Accessed","author":"Michelangelo Uber","year":"2021","unstructured":"Uber Michelangelo , https:\/\/eng.uber.com\/michelangelo\/ , Accessed March 22, 2021 . Uber Michelangelo, https:\/\/eng.uber.com\/michelangelo\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_39_1","volume-title":"Accessed","author":"Learner Flow Facebook's","year":"2021","unstructured":"Facebook's FB Learner Flow , https:\/\/engineering.fb.com\/core-data\/introducing-fblearner-flow-facebook-s-ai-backbone\/ , Accessed March 22, 2021 . Facebook's FBLearner Flow, https:\/\/engineering.fb.com\/core-data\/introducing-fblearner-flow-facebook-s-ai-backbone\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_40_1","volume-title":"Accessed","year":"2021","unstructured":"H2o.AI , https:\/\/www.h2o.ai\/ , Accessed March 22, 2021 . H2o.AI, https:\/\/www.h2o.ai\/, Accessed March 22, 2021."},{"key":"e_1_3_2_2_41_1","volume-title":"Automated Sanity Checking for ML Data Sets. In NIPS MLSys Workshop","author":"Hynes Nick","year":"2017","unstructured":"Nick Hynes , D Sculley , and Michael Terry . The Data Linter: Lightweight , Automated Sanity Checking for ML Data Sets. In NIPS MLSys Workshop , 2017 . Nick Hynes, D Sculley, and Michael Terry. The Data Linter: Lightweight, Automated Sanity Checking for ML Data Sets. In NIPS MLSys Workshop, 2017."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/1925844.1926423"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2240236.2240260"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.14778\/3231751.3231766"},{"key":"e_1_3_2_2_45_1","first-page":"2117","volume-title":"Proceedings of the 2016 International Conference on Management of Data, SIGMOD Conference 2016","author":"Krishnan Sanjay","year":"2016","unstructured":"Sanjay Krishnan , Michael J. Franklin , Ken Goldberg , Jiannan Wang , and Eugene Wu. ActiveClean : An Interactive Data Cleaning Framework For Modern Machine Learning. In Fatma \u00d6 zcan, Georgia Koutrika, and Sam Madden, editors , Proceedings of the 2016 International Conference on Management of Data, SIGMOD Conference 2016 , San Francisco, CA, USA, June 26 - July 01, 2016 , pages 2117 -- 2120 . ACM, 2016. Sanjay Krishnan, Michael J. Franklin, Ken Goldberg, Jiannan Wang, and Eugene Wu. ActiveClean: An Interactive Data Cleaning Framework For Modern Machine Learning. In Fatma \u00d6 zcan, Georgia Koutrika, and Sam Madden, editors, Proceedings of the 2016 International Conference on Management of Data, SIGMOD Conference 2016, San Francisco, CA, USA, June 26 - July 01, 2016, pages 2117--2120. ACM, 2016."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.14778\/3229863.3229867"},{"key":"e_1_3_2_2_47_1","volume-title":"Accessed","author":"BigQuery Schema Detection","year":"2021","unstructured":"Schema Detection BigQuery , https:\/\/cloud.google.com\/bigquery\/docs\/schema-detect , Accessed March 22, 2021 . Schema Detection BigQuery, https:\/\/cloud.google.com\/bigquery\/docs\/schema-detect, Accessed March 22, 2021."},{"key":"e_1_3_2_2_48_1","volume-title":"Extending Database Technology (EDBT)","author":"Baazizi Mohamed-Amine","year":"2017","unstructured":"Mohamed-Amine Baazizi , Houssem Ben Lahmar , Dario Colazzo , Giorgio Ghelli , and Carlo Sartiani . Schema Inference for Massive JSON Datasets . In Extending Database Technology (EDBT) , 2017 . Mohamed-Amine Baazizi, Houssem Ben Lahmar, Dario Colazzo, Giorgio Ghelli, and Carlo Sartiani. Schema Inference for Massive JSON Datasets. In Extending Database Technology (EDBT), 2017."},{"key":"e_1_3_2_2_49_1","volume-title":"An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909","author":"Gijsbers Pieter","year":"2019","unstructured":"Pieter Gijsbers , Erin LeDell , Janek Thomas , S\u00e9bastien Poirier , Bernd Bischl , and Joaquin Vanschoren . An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909 , 2019 . Pieter Gijsbers, Erin LeDell, Janek Thomas, S\u00e9bastien Poirier, Bernd Bischl, and Joaquin Vanschoren. An Open Source AutoML Benchmark. arXiv preprint arXiv:1907.00909, 2019."},{"key":"e_1_3_2_2_50_1","volume-title":"CleanML: A Benchmark for Joint Data Cleaning and Machine Learning [Experiments and Analysis]. arXiv preprint arXiv:1904.09483","author":"Li Peng","year":"2019","unstructured":"Peng Li , Xi Rao , Jennifer Blase , Yue Zhang , Xu Chu , and Ce Zhang . CleanML: A Benchmark for Joint Data Cleaning and Machine Learning [Experiments and Analysis]. arXiv preprint arXiv:1904.09483 , 2019 . Peng Li, Xi Rao, Jennifer Blase, Yue Zhang, Xu Chu, and Ce Zhang. CleanML: A Benchmark for Joint Data Cleaning and Machine Learning [Experiments and Analysis]. arXiv preprint arXiv:1904.09483, 2019."}],"event":{"name":"SIGMOD\/PODS '21: International Conference on Management of Data","location":"Virtual Event China","acronym":"SIGMOD\/PODS '21","sponsor":["SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 2021 International Conference on Management of Data"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3448016.3457274","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3448016.3457274","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3448016.3457274","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:28:06Z","timestamp":1750195686000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3448016.3457274"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,9]]},"references-count":50,"alternative-id":["10.1145\/3448016.3457274","10.1145\/3448016"],"URL":"https:\/\/doi.org\/10.1145\/3448016.3457274","relation":{},"subject":[],"published":{"date-parts":[[2021,6,9]]},"assertion":[{"value":"2021-06-18","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}