{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T20:14:45Z","timestamp":1777407285124,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":109,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,27]],"date-time":"2024-10-27T00:00:00Z","timestamp":1729987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,27]]},"DOI":"10.1145\/3691620.3695267","type":"proceedings-article","created":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T15:39:19Z","timestamp":1729265959000},"page":"2041-2052","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["AutoDW: Automatic Data Wrangling Leveraging Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2850-2048","authenticated-orcid":false,"given":"Lei","family":"Liu","sequence":"first","affiliation":[{"name":"Fujitsu Research of America Inc., Santa Clara, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2239-1014","authenticated-orcid":false,"given":"So","family":"Hasegawa","sequence":"additional","affiliation":[{"name":"Fujitsu Research of America Inc., Santa Clara, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9873-3342","authenticated-orcid":false,"given":"Shailaja Keyur","family":"Sampat","sequence":"additional","affiliation":[{"name":"Fujitsu Research of America Inc., Santa Clara, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5064-0813","authenticated-orcid":false,"given":"Maria","family":"Xenochristou","sequence":"additional","affiliation":[{"name":"Fujitsu Research of America Inc., Santa Clara, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4351-7415","authenticated-orcid":false,"given":"Wei-Peng","family":"Chen","sequence":"additional","affiliation":[{"name":"Fujitsu Research of America Inc., Santa Clara, California, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0097-5923","authenticated-orcid":false,"given":"Takashi","family":"Kato","sequence":"additional","affiliation":[{"name":"Fujitsu Research, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7870-0719","authenticated-orcid":false,"given":"Taisei","family":"Kakibuchi","sequence":"additional","affiliation":[{"name":"Fujitsu Research, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9708-2929","authenticated-orcid":false,"given":"Tatsuya","family":"Asai","sequence":"additional","affiliation":[{"name":"Fujitsu Research, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"accessed","author":"Market Size Data Wrangling","year":"2024","unstructured":"Data Wrangling Market Size & Share Analysis Growth Trends & Forecasts (2024 2029). accessed July 02, 2024. https:\/\/www.mordorintelligence.com\/industry-reports\/data-wrangling-market"},{"key":"e_1_3_2_1_2_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al. 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_3_1","volume-title":"accessed","year":"2024","unstructured":"all-mpnet-base v2. accessed July 10, 2024. https:\/\/huggingface.co\/sentence-transformers\/all-mpnet-base-v2"},{"key":"e_1_3_2_1_4_1","volume-title":"accessed","year":"2024","unstructured":"Alteryx. accessed July 10, 2024. https:\/\/www.alteryx.com\/"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098021"},{"key":"e_1_3_2_1_6_1","volume-title":"Principles and procedures of exploratory data analysis. Psychological methods 2, 2","author":"Behrens John T","year":"1997","unstructured":"John T Behrens. 1997. Principles and procedures of exploratory data analysis. Psychological methods 2, 2 (1997), 131."},{"key":"e_1_3_2_1_7_1","volume-title":"Data wrangling with R","author":"Boehmke Bradley C","unstructured":"Bradley C Boehmke. 2016. Data wrangling with R. Springer."},{"key":"e_1_3_2_1_8_1","volume-title":"Random forests. Machine learning 45","author":"Breiman Leo","year":"2001","unstructured":"Leo Breiman. 2001. Random forests. Machine learning 45 (2001), 5--32."},{"key":"e_1_3_2_1_9_1","volume-title":"Introduction to time series and forecasting","author":"Brockwell Peter J","unstructured":"Peter J Brockwell and Richard A Davis. 2002. Introduction to time series and forecasting. Springer."},{"key":"e_1_3_2_1_10_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_11_1","volume-title":"The effects of data quality on machine learning performance. arXiv preprint arXiv:2207.14529","author":"Budach Lukas","year":"2022","unstructured":"Lukas Budach, Moritz Feuerpfeil, Nina Ihde, Andrea Nathansen, Nele Noack, Hendrik Patzlaff, Felix Naumann, and Hazar Harmouch. 2022. The effects of data quality on machine learning performance. arXiv preprint arXiv:2207.14529 (2022)."},{"key":"e_1_3_2_1_12_1","volume-title":"The challenges of data quality and data quality assessment in the big data era. Data science journal 14","author":"Cai Li","year":"2015","unstructured":"Li Cai and Yangyong Zhu. 2015. The challenges of data quality and data quality assessment in the big data era. Data science journal 14 (2015), 2--2."},{"key":"e_1_3_2_1_13_1","volume-title":"Jumping NLP curves: A review of natural language processing research","author":"Cambria Erik","year":"2014","unstructured":"Erik Cambria and Bebo White. 2014. Jumping NLP curves: A review of natural language processing research. IEEE Computational intelligence magazine 9, 2 (2014), 48--57."},{"key":"e_1_3_2_1_14_1","volume-title":"How Do Large Language Models Acquire Factual Knowledge During Pretraining? arXiv preprint arXiv:2406.11813","author":"Chang Hoyeon","year":"2024","unstructured":"Hoyeon Chang, Jinho Park, Seonghyeon Ye, Sohee Yang, Youngkyung Seo, Du-Seong Chang, and Minjoon Seo. 2024. How Do Large Language Models Acquire Factual Knowledge During Pretraining? arXiv preprint arXiv:2406.11813 (2024)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447556.3447567"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2882903.2912574"},{"key":"e_1_3_2_1_17_1","volume-title":"accessed","year":"2024","unstructured":"Cohere. accessed September 12, 2024. https:\/\/cohere.com\/"},{"key":"e_1_3_2_1_18_1","volume-title":"accessed","author":"Configuration ML","year":"2024","unstructured":"SapientML Configuration. accessed July 09, 2024. https:\/\/sapientml.readthedocs.io\/en\/latest\/user\/configuration.html#parameters-for-sapientml"},{"key":"e_1_3_2_1_19_1","volume-title":"accessed","year":"2024","unstructured":"dataprep. accessed July 10, 2024. https:\/\/dataprep.ai\/"},{"key":"e_1_3_2_1_20_1","volume-title":"accessed","year":"2024","unstructured":"DataRobot. accessed July 10, 2024. https:\/\/www.datarobot.com\/"},{"key":"e_1_3_2_1_21_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"accessed","author":"BERT.","year":"2024","unstructured":"DistilBERT. accessed July 10, 2024. https:\/\/huggingface.co\/distilbert\/distilbert-base-multilingual-cased"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1080\/00031305.1976.10479172"},{"key":"e_1_3_2_1_24_1","volume-title":"accessed","year":"2024","unstructured":"dotData. accessed July 10, 2024. https:\/\/dotdata.com\/"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2015.05.197"},{"key":"e_1_3_2_1_26_1","volume-title":"Autogluon-tabular: Robust and accurate automl for structured data. arXiv preprint arXiv:2003.06505","author":"Erickson Nick","year":"2020","unstructured":"Nick Erickson, Jonas Mueller, Alexander Shirkov, Hang Zhang, Pedro Larroy, Mu Li, and Alexander Smola. 2020. Autogluon-tabular: Robust and accurate automl for structured data. arXiv preprint arXiv:2003.06505 (2020)."},{"key":"e_1_3_2_1_27_1","volume-title":"accessed","author":"Factory Azure Data","year":"2024","unstructured":"Azure Data Factory. accessed July 10, 2024. https:\/\/azure.microsoft.com\/en-us\/products\/data-factory"},{"key":"e_1_3_2_1_28_1","volume-title":"accessed","author":"Streamlit A","year":"2024","unstructured":"Streamlit A faster way to build and share data apps. accessed July 10, 2024. https:\/\/streamlit.io\/"},{"key":"e_1_3_2_1_29_1","volume-title":"Efficient and robust automated machine learning. Advances in neural information processing systems 28","author":"Feurer Matthias","year":"2015","unstructured":"Matthias Feurer, Aaron Klein, Katharina Eggensperger, Jost Springenberg, Manuel Blum, and Frank Hutter. 2015. Efficient and robust automated machine learning. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_2_1_30_1","volume-title":"accessed","author":"Functions Azure","year":"2024","unstructured":"Azure Functions. accessed July 08, 2024. https:\/\/azure.microsoft.com\/en-us\/products\/functions"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2007.1015"},{"key":"e_1_3_2_1_32_1","volume-title":"19th International Conference on Extending Database Technology. 473--478","author":"Furche Tim","year":"2016","unstructured":"Tim Furche, Georg Gottlob, Leonid Libkin, Giorgio Orsi, and Norman W Paton. 2016. Data wrangling for big data: Challenges and opportunities. In 19th International Conference on Extending Database Technology. 473--478."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.295"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-27520-4_13"},{"key":"e_1_3_2_1_35_1","volume-title":"accessed","year":"2024","unstructured":"GPT-4o. accessed July 10, 2024. https:\/\/openai.com\/index\/hello-gpt-4o\/"},{"key":"e_1_3_2_1_36_1","first-page":"1","article-title":"Data quality considerations for big data and machine learning: Going beyond data cleaning and transformations","volume":"10","author":"Gudivada Venkat","year":"2017","unstructured":"Venkat Gudivada, Amy Apon, and Junhua Ding. 2017. Data quality considerations for big data and machine learning: Going beyond data cleaning and transformations. International Journal on Advances in Software 10, 1 (2017), 1--20.","journal-title":"International Journal on Advances in Software"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3470817"},{"key":"e_1_3_2_1_38_1","volume-title":"Naveed Akhtar, Jia Wu, Seyedali Mirjalili, et al.","author":"Hadi Muhammad Usman","year":"2023","unstructured":"Muhammad Usman Hadi, Rizwan Qureshi, Abbas Shah, Muhammad Irfan, Anas Zafar, Muhammad Bilal Shaikh, Naveed Akhtar, Jia Wu, Seyedali Mirjalili, et al. 2023. A survey on large language models: Applications, challenges, limitations, and practical usage. Authorea Preprints (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"R for data science: import, tidy, transform, visualize, and model data","author":"Hadley Wickham","year":"2016","unstructured":"Wickham Hadley and Grolemund Garrett. 2016. R for data science: import, tidy, transform, visualize, and model data. O'Reilly Media, Inc (2016)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-1119"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106622"},{"key":"e_1_3_2_1_42_1","volume-title":"Large language models for automated data science: Introducing caafe for context-aware automated feature engineering. Advances in Neural Information Processing Systems 36","author":"Hollmann Noah","year":"2024","unstructured":"Noah Hollmann, Samuel M\u00fcller, and Frank Hutter. 2024. Large language models for automated data science: Introducing caafe for context-aware automated feature engineering. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406477"},{"key":"e_1_3_2_1_44_1","volume-title":"LLM-Select: Feature Selection with Large Language Models. arXiv preprint arXiv:2407.02694","author":"Jeong Daniel P","year":"2024","unstructured":"Daniel P Jeong, Zachary C Lipton, and Pradeep Ravikumar. 2024. LLM-Select: Feature Selection with Large Language Models. arXiv preprint arXiv:2407.02694 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"accessed","year":"2024","unstructured":"Jinja. accessed July 10, 2024. https:\/\/jinja.palletsprojects.com\/en\/3.1.x\/"},{"key":"e_1_3_2_1_46_1","unstructured":"Richard Arnold Johnson Dean W Wichern et al. 2002. Applied multivariate statistical analysis. (2002)."},{"key":"e_1_3_2_1_47_1","volume-title":"Principal component analysis for special types of data","author":"Jolliffe Ian T","unstructured":"Ian T Jolliffe. 2002. Principal component analysis for special types of data. Springer."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2014.01.003"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3470918"},{"key":"e_1_3_2_1_50_1","volume-title":"Garnett (Eds.)","volume":"30","author":"Ke Guolin","year":"2017","unstructured":"Guolin Ke, Qi Meng, Thomas Finley, Taifeng Wang, Wei Chen, Weidong Ma, Qiwei Ye, and Tie-Yan Liu. 2017. LightGBM: A Highly Efficient Gradient Boosting Decision Tree. In Advances in Neural Information Processing Systems, I. Guyon, U. Von Luxburg, S. Bengio, H. Wallach, R. Fergus, S. Vishwanathan, and R. Garnett (Eds.), Vol. 30. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2017\/file\/6449f44a102fde848669bdd9eb6b76fa-Paper.pdf"},{"key":"e_1_3_2_1_51_1","volume-title":"Your Machine Learning and Data Science Community. accessed","author":"Kaggle","year":"2024","unstructured":"Kaggle: Your Machine Learning and Data Science Community. accessed July 11, 2024. https:\/\/www.kaggle.com\/"},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the AutoML Workshop at ICML","volume":"2020","author":"LeDell Erin","year":"2020","unstructured":"Erin LeDell and Sebastien Poirier. 2020. H2o automl: Scalable automatic machine learning. In Proceedings of the AutoML Workshop at ICML, Vol. 2020. ICML San Diego, CA, USA."},{"key":"e_1_3_2_1_53_1","volume-title":"The Internet of Things (IoT): Applications, investments, and challenges for enterprises. Business horizons 58, 4","author":"Lee In","year":"2015","unstructured":"In Lee and Kyoochun Lee. 2015. The Internet of Things (IoT): Applications, investments, and challenges for enterprises. Business horizons 58, 4 (2015), 431--440."},{"key":"e_1_3_2_1_54_1","volume-title":"Wayne Xin Zhao, and Ji-Rong Wen","author":"Li Junyi","year":"2021","unstructured":"Junyi Li, Tianyi Tang, Wayne Xin Zhao, and Ji-Rong Wen. 2021. Pretrained language models for text generation: A survey. arXiv preprint arXiv:2105.10311 (2021)."},{"key":"e_1_3_2_1_55_1","volume-title":"2023 International Conference on Learning Representations (ICLR).","author":"Li Liyao","year":"2023","unstructured":"Liyao Li, Haobo Wang, Liangyu Zha, Qingyi Huang, Sai Wu, Gang Chen, and Junbo Zhao. 2023. Learning a Data-Driven Policy Network for Pre-Training Automated Feature Engineering. In 2023 International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_56_1","first-page":"1","article-title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","volume":"55","author":"Liu Pengfei","year":"2023","unstructured":"Pengfei Liu, Weizhe Yuan, Jinlan Fu, Zhengbao Jiang, Hiroaki Hayashi, and Graham Neubig. 2023. Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing. Comput. Surveys 55, 9 (2023), 1--35.","journal-title":"Comput. Surveys"},{"key":"e_1_3_2_1_57_1","volume-title":"GPT understands, too. arXiv preprint arXiv:2103.10385","author":"Liu Xiao","year":"2021","unstructured":"Xiao Liu, Yanan Zheng, Zhengxiao Du, Ming Ding, Yujie Qian, Zhilin Yang, and Jie Tang. 2021. GPT understands, too. arXiv preprint arXiv:2103.10385 (2021)."},{"key":"e_1_3_2_1_58_1","volume-title":"Automated machine learning for structured data. accessed","year":"2024","unstructured":"TransmogrifAI: Automated machine learning for structured data. accessed July 10, 2024. https:\/\/transmogrif.ai\/"},{"key":"e_1_3_2_1_59_1","volume-title":"International conference on data intelligence and cognitive informatics. Springer, 387--402","author":"Marvin Ggaliwango","year":"2023","unstructured":"Ggaliwango Marvin, Nakayiza Hellen, Daudi Jjingo, and Joyce Nakatumba-Nabende. 2023. Prompt engineering in large language models. In International conference on data intelligence and cognitive informatics. Springer, 387--402."},{"key":"e_1_3_2_1_60_1","volume-title":"Python for data analysis. \" O'Reilly Media","author":"McKinney Wes","unstructured":"Wes McKinney. 2013. Python for data analysis. \" O'Reilly Media, Inc.\"."},{"key":"e_1_3_2_1_61_1","first-page":"51","article-title":"Data structures for statistical computing in Python","volume":"445","author":"McKinney Wes","year":"2010","unstructured":"Wes McKinney et al. 2010. Data structures for statistical computing in Python.. In SciPy, Vol. 445. 51--56.","journal-title":"SciPy"},{"key":"e_1_3_2_1_62_1","unstructured":"Wes McKinney et al. 2011. pandas: a foundational Python library for data analysis and statistics. Python for high performance and scientific computing 14 9 (2011) 1--9."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE51399.2021.00013"},{"key":"e_1_3_2_1_64_1","volume-title":"Can Foundation Models Wrangle Your Data? arXiv preprint arXiv:2205.09911","author":"Narayan Avanika","year":"2022","unstructured":"Avanika Narayan, Ines Chami, Laurel Orr, and Christopher R\u00e9. 2022. Can Foundation Models Wrangle Your Data? arXiv preprint arXiv:2205.09911 (2022)."},{"key":"e_1_3_2_1_65_1","volume-title":"Business intelligence. Handbook on decision support systems 2","author":"Negash Solomon","year":"2008","unstructured":"Solomon Negash and Paul Gray. 2008. Business intelligence. Handbook on decision support systems 2 (2008), 175--193."},{"key":"e_1_3_2_1_66_1","volume-title":"accessed","author":"Notebook Jupyter","year":"2024","unstructured":"Jupyter Notebook. accessed July 10, 2024. https:\/\/jupyter.org\/"},{"key":"e_1_3_2_1_67_1","volume-title":"accessed","author":"Data A Grammar","year":"2024","unstructured":"A Grammar of Data Manipulation dplyr. accessed July 10, 2024. https:\/\/dplyr.tidyverse.org\/"},{"key":"e_1_3_2_1_68_1","volume-title":"Workshop on automatic machine learning. PMLR, 66--74","author":"Olson Randal S","year":"2016","unstructured":"Randal S Olson and Jason H Moore. 2016. TPOT: A tree-based pipeline optimization tool for automating machine learning. In Workshop on automatic machine learning. PMLR, 66--74."},{"key":"e_1_3_2_1_69_1","volume-title":"accessed","author":"ML.","year":"2024","unstructured":"OpenML. accessed July 10, 2024. https:\/\/www.openml.org\/"},{"key":"e_1_3_2_1_70_1","volume-title":"accessed","author":"Analysis Library Python Data","year":"2024","unstructured":"pandas Python Data Analysis Library. accessed July 10, 2024. https:\/\/pandas.pydata.org\/"},{"key":"e_1_3_2_1_71_1","volume-title":"accessed","year":"2024","unstructured":"paraphrase-multilingual-mpnet-base v2. accessed July 10, 2024. https:\/\/huggingface.co\/sentence-transformers\/paraphrase-multilingual-mpnet-base-v2"},{"key":"e_1_3_2_1_72_1","volume-title":"21st International Workshop on Design, Optimization, Languages and Analytical Processing of Big Data.","author":"Paton Norman","year":"2019","unstructured":"Norman Paton. 2019. Automating data preparation: Can we? should we? must we?. In 21st International Workshop on Design, Optimization, Languages and Analytical Processing of Big Data."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1250"},{"key":"e_1_3_2_1_74_1","volume-title":"Text clustering with LLM embeddings. arXiv preprint arXiv:2403.15112","author":"Petukhova Alina","year":"2024","unstructured":"Alina Petukhova, Joao P Matos-Carvalho, and Nuno Fachada. 2024. Text clustering with LLM embeddings. arXiv preprint arXiv:2403.15112 (2024)."},{"key":"e_1_3_2_1_75_1","volume-title":"MLJAR: State-of-the-art Automated Machine Learning Framework for Tabular Data. https:\/\/github.com\/mljar\/mljar-supervised","author":"P\u0142o\u0144ska Aleksandra","year":"2021","unstructured":"Aleksandra P\u0142o\u0144ska and Piotr P\u0142o\u0144ski. 2021. MLJAR: State-of-the-art Automated Machine Learning Framework for Tabular Data. https:\/\/github.com\/mljar\/mljar-supervised"},{"key":"e_1_3_2_1_76_1","volume-title":"Data science and its relationship to big data and data-driven decision making. Big data 1, 1","author":"Provost Foster","year":"2013","unstructured":"Foster Provost and Tom Fawcett. 2013. Data science and its relationship to big data and data-driven decision making. Big data 1, 1 (2013), 51--59."},{"key":"e_1_3_2_1_77_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog 1 8 (2019) 9."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.5555\/3455716.3455856"},{"key":"e_1_3_2_1_79_1","volume-title":"Principles of data wrangling: Practical techniques for data preparation. \" O'Reilly Media","author":"Rattenbury Tye","unstructured":"Tye Rattenbury, Joseph M Hellerstein, Jeffrey Heer, Sean Kandel, and Connor Carreras. 2017. Principles of data wrangling: Practical techniques for data preparation. \" O'Reilly Media, Inc.\"."},{"key":"e_1_3_2_1_80_1","volume-title":"accessed","author":"Azure","year":"2024","unstructured":"Azure OpenAI Service REST API reference. accessed July 10, 2024. https:\/\/learn.microsoft.com\/en-us\/azure\/ai-services\/openai\/reference"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_82_1","volume-title":"How much knowledge can you pack into the parameters of a language model? arXiv preprint arXiv:2002.08910","author":"Roberts Adam","year":"2020","unstructured":"Adam Roberts, Colin Raffel, and Noam Shazeer. 2020. How much knowledge can you pack into the parameters of a language model? arXiv preprint arXiv:2002.08910 (2020)."},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1145\/3510003.3510226"},{"key":"e_1_3_2_1_84_1","volume-title":"Sriparna Saha, Vinija Jain, Samrat Mondal, and Aman Chadha.","author":"Sahoo Pranab","year":"2024","unstructured":"Pranab Sahoo, Ayush Kumar Singh, Sriparna Saha, Vinija Jain, Samrat Mondal, and Aman Chadha. 2024. A systematic survey of prompt engineering in large language models: Techniques and applications. arXiv preprint arXiv:2402.07927 (2024)."},{"key":"e_1_3_2_1_85_1","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2020. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. arXiv:cs.CL\/1910.01108 https:\/\/arxiv.org\/abs\/1910.01108"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.185"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"},{"key":"e_1_3_2_1_88_1","volume-title":"accessed","year":"2024","unstructured":"SentenceTransformers. accessed July 10, 2024. https:\/\/sbert.net\/"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/3448016.3457274"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1145\/3448016.3457274"},{"key":"e_1_3_2_1_91_1","volume-title":"Vu Le, Chris Parnin, Mukul Singh, and Gust Verbruggen.","author":"Singha Ananya","year":"2024","unstructured":"Ananya Singha, Bhavya Chopra, Anirudh Khatry, Sumit Gulwani, Austin Z Henley, Vu Le, Chris Parnin, Mukul Singh, and Gust Verbruggen. 2024. Semantically Aligned Question and Code Generation for Automated Insight Generation. arXiv preprint arXiv:2405.01556 (2024)."},{"key":"e_1_3_2_1_92_1","doi-asserted-by":"publisher","DOI":"10.1108\/eb026526"},{"key":"e_1_3_2_1_93_1","volume-title":"accessed","author":"Specification JSON-RPC","year":"2024","unstructured":"JSON-RPC 2.0 Specification. accessed July 08, 2024. https:\/\/www.jsonrpc.org\/specification"},{"key":"e_1_3_2_1_94_1","unstructured":"Yu Sun Shuohuan Wang Shikun Feng Siyu Ding Chao Pang Junyuan Shang Jiaxiang Liu Xuyi Chen Yanbin Zhao Yuxiang Lu et al. 2021. Ernie 3.0: Large-scale knowledge enhanced pre-training for language understanding and generation. arXiv preprint arXiv:2107.02137 (2021)."},{"key":"e_1_3_2_1_95_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Yonghui Wu Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_96_1","volume-title":"accessed","author":"Data Science Overcoming","year":"2024","unstructured":"Overcoming the 80\/20 Rule in Data Science. accessed July 10, 2024. https:\/\/www.pragmaticinstitute.com\/resources\/articles\/data\/overcoming-the-80-20-rule-in-data-science\/"},{"key":"e_1_3_2_1_97_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al. 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_98_1","volume-title":"accessed","author":"Models SentenceTransformers","year":"2024","unstructured":"SentenceTransformers Pre trained Models. accessed July 10, 2024. https:\/\/sbert.net\/docs\/sentence_transformer\/pretrained_models.html"},{"key":"e_1_3_2_1_99_1","unstructured":"John Wilder Tukey et al. 1977. Exploratory data analysis. Vol. 2. Springer."},{"key":"e_1_3_2_1_100_1","volume-title":"Python data science handbook: Essential tools for working with data. \" O'Reilly Media","author":"VanderPlas Jake","unstructured":"Jake VanderPlas. 2016. Python data science handbook: Essential tools for working with data. \" O'Reilly Media, Inc.\"."},{"key":"e_1_3_2_1_101_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_102_1","first-page":"434","article-title":"FLAML: A fast and lightweight automl library","volume":"3","author":"Wang Chi","year":"2021","unstructured":"Chi Wang, Qingyun Wu, Markus Weimer, and Erkang Zhu. 2021. FLAML: A fast and lightweight automl library. Proceedings of Machine Learning and Systems 3 (2021), 434--447.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_103_1","volume-title":"Improving text embeddings with large language models. arXiv preprint arXiv:2401.00368","author":"Wang Liang","year":"2023","unstructured":"Liang Wang, Nan Yang, Xiaolong Huang, Linjun Yang, Rangan Majumder, and Furu Wei. 2023. Improving text embeddings with large language models. arXiv preprint arXiv:2401.00368 (2023)."},{"key":"e_1_3_2_1_104_1","volume-title":"accessed","author":"SageMaker Data Wrangler Amazon","year":"2024","unstructured":"Amazon SageMaker Data Wrangler. accessed July 10, 2024. https:\/\/aws.amazon.com\/sagemaker\/data-wrangler\/"},{"key":"e_1_3_2_1_105_1","volume-title":"Soumi Das, Vedant Nanda, Bishwamittra Ghosh, Camila Kolling, Till Speicher, Laurent Bindschaedler, Krishna P Gummadi, and Evimaria Terzi.","author":"Wu Qinyuan","year":"2024","unstructured":"Qinyuan Wu, Mohammad Aflah Khan, Soumi Das, Vedant Nanda, Bishwamittra Ghosh, Camila Kolling, Till Speicher, Laurent Bindschaedler, Krishna P Gummadi, and Evimaria Terzi. 2024. Towards Reliable Latent Knowledge Estimation in LLMs: In-Context Learning vs. Prompting Based Factual Knowledge Extraction. arXiv preprint arXiv:2404.12957 (2024)."},{"key":"e_1_3_2_1_106_1","volume-title":"Harnessing the power of llms in practice: A survey on chatgpt and beyond. ACM Transactions on Knowledge Discovery from Data 18, 6","author":"Yang Jingfeng","year":"2024","unstructured":"Jingfeng Yang, Hongye Jin, Ruixiang Tang, Xiaotian Han, Qizhang Feng, Haoming Jiang, Shaochen Zhong, Bing Yin, and Xia Hu. 2024. Harnessing the power of llms in practice: A survey on chatgpt and beyond. ACM Transactions on Knowledge Discovery from Data 18, 6 (2024), 1--32."},{"key":"e_1_3_2_1_107_1","volume-title":"Dynamic and Adaptive Feature Generation with LLM. arXiv preprint arXiv:2406.03505","author":"Zhang Xinhao","year":"2024","unstructured":"Xinhao Zhang, Jinghan Zhang, Banafsheh Rekabdar, Yuanchun Zhou, Pengfei Wang, and Kunpeng Liu. 2024. Dynamic and Adaptive Feature Generation with LLM. arXiv preprint arXiv:2406.03505 (2024)."},{"key":"e_1_3_2_1_108_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A survey of large language models. arXiv preprint arXiv:2303.18223 (2023)."},{"key":"e_1_3_2_1_109_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6523"}],"event":{"name":"ASE '24: 39th IEEE\/ACM International Conference on Automated Software Engineering","location":"Sacramento CA USA","acronym":"ASE '24","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS"]},"container-title":["Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695267","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3691620.3695267","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:07Z","timestamp":1750291447000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695267"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,27]]},"references-count":109,"alternative-id":["10.1145\/3691620.3695267","10.1145\/3691620"],"URL":"https:\/\/doi.org\/10.1145\/3691620.3695267","relation":{},"subject":[],"published":{"date-parts":[[2024,10,27]]},"assertion":[{"value":"2024-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}