{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T03:16:46Z","timestamp":1783135006071,"version":"3.54.6"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100019069","name":"Chinese Academy of Engineering","doi-asserted-by":"publisher","award":["2024-JZ-0301"],"award-info":[{"award-number":["2024-JZ-0301"]}],"id":[{"id":"10.13039\/501100019069","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012165","name":"Key Technologies Research and Development Program","doi-asserted-by":"publisher","award":["2023YFC3305200"],"award-info":[{"award-number":["2023YFC3305200"]}],"id":[{"id":"10.13039\/501100012165","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai Municipality","doi-asserted-by":"publisher","award":["23ZR1404900"],"award-info":[{"award-number":["23ZR1404900"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.eswa.2026.132645","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:53:02Z","timestamp":1777571582000},"page":"132645","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["CATS : An enhanced framework in LLM-based tabular data synthesis by correlation augmentation"],"prefix":"10.1016","volume":"325","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-2372-4022","authenticated-orcid":false,"given":"Luyu","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingxuan","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyue","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2230-7671","authenticated-orcid":false,"given":"Sen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongfeng","family":"Chai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132645_bib0001","unstructured":"AlKhamissi, B., Li, M., Celikyilmaz, A., Diab, M., & Ghazvininejad, M. (2022). A review on language models as knowledge bases. arXiv: 2204.06031."},{"key":"10.1016\/j.eswa.2026.132645_bib0002","doi-asserted-by":"crossref","first-page":"1329","DOI":"10.1007\/s10994-019-05791-5","article-title":"Data scarcity, robustness and extreme multi-label classification","volume":"108","author":"Babbar","year":"2019","journal-title":"Machine Learning"},{"key":"10.1016\/j.eswa.2026.132645_bib0003","series-title":"The eleventh international conference on learning representations","article-title":"Language models are realistic tabular data generators","author":"Borisov","year":"2023"},{"key":"10.1016\/j.eswa.2026.132645_bib0004","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"3","key":"10.1016\/j.eswa.2026.132645_bib0005","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3502289","article-title":"Ai in finance: Challenges, techniques, and opportunities","volume":"55","author":"Cao","year":"2022","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"10.1016\/j.eswa.2026.132645_bib0006","series-title":"Findings of the association for computational linguistics: EACL 2023","first-page":"1856","article-title":"Crawling the internal knowledge-base of language models","author":"Cohen","year":"2023"},{"issue":"3","key":"10.1016\/j.eswa.2026.132645_bib0007","doi-asserted-by":"crossref","first-page":"220","DOI":"10.1038\/s42256-023-00626-4","article-title":"Parameter-efficient fine-tuning of large-scale pre-trained language models","volume":"5","author":"Ding","year":"2023","journal-title":"Nature Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132645_bib0008","doi-asserted-by":"crossref","DOI":"10.3389\/frai.2022.826737","article-title":"Ai technologies, privacy, and security","volume":"5","author":"Elliott","year":"2022","journal-title":"Frontiers in Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.132645_bib0009","article-title":"Large language models (LLMs) on tabular data: prediction, generation, and understanding - a survey","author":"Fang","year":"2024","journal-title":"Transactions on Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.132645_bib0010","doi-asserted-by":"crossref","unstructured":"Gao, J., Du, Z., Li, X., Zhao, X., Wang, Y., Li, X., Guo, H., & Tang, R. (2025). SampleLLM: Optimizing tabular data synthesis in recommendations. arXiv: 2501.16125.","DOI":"10.32388\/A9U1SH"},{"key":"10.1016\/j.eswa.2026.132645_bib0011","doi-asserted-by":"crossref","DOI":"10.1073\/pnas.2305016120","article-title":"ChatGPT outperforms crowd workers for text-annotation tasks","volume":"120","author":"Gilardi","year":"2023","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"10.1016\/j.eswa.2026.132645_bib0012","series-title":"International conference on learning representations","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132645_bib0013","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125852","article-title":"LLMoverTab: Tabular data augmentation with language model-driven oversampling","volume":"264","author":"Isomura","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132645_bib0014","unstructured":"Kadavath, S., Conerly, T., Askell, A., Henighan, T., Drain, D., Perez, E., Schiefer, N., Hatfield-Dodds, Z., DasSarma, N., Tran-Johnson, E. et al. (2022). Language models (mostly) know what they know. arXiv: 2207.05221."},{"key":"10.1016\/j.eswa.2026.132645_bib0015","series-title":"International conference on machine learning","first-page":"17564","article-title":"TabDDPM: Modelling tabular data with diffusion models","author":"Kotelnikov","year":"2023"},{"key":"10.1016\/j.eswa.2026.132645_bib0016","series-title":"Forty-first international conference on machine learning","article-title":"Cllms: Consistency large language models","author":"Kou","year":"2024"},{"key":"10.1016\/j.eswa.2026.132645_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112438","article-title":"TabSAL: Synthesizing tabular data with small agent assisted language models","volume":"304","author":"Li","year":"2024","journal-title":"Knowledge-Based Systems"},{"issue":"3","key":"10.1016\/j.eswa.2026.132645_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.ipm.2024.103645","article-title":"End-to-end approach of multi-grained embedding of categorical features in tabular data","volume":"61","author":"Liu","year":"2024","journal-title":"Information Processing & Management"},{"key":"10.1016\/j.eswa.2026.132645_bib0019","unstructured":"Liu, R., Wei, J., Liu, F., Si, C., Zhang, Y., Rao, J., Zheng, S., Peng, D., Yang, D., Zhou, D. et al. (2024b). Best practices and lessons learned on synthetic data for language models. arXiv: 2404.07503."},{"issue":"2","key":"10.1016\/j.eswa.2026.132645_bib0020","doi-asserted-by":"crossref","first-page":"143","DOI":"10.11613\/BM.2013.018","article-title":"The chi-square test of independence","volume":"23","author":"McHugh","year":"2013","journal-title":"Biochemia Medica"},{"key":"10.1016\/j.eswa.2026.132645_bib0021","doi-asserted-by":"crossref","first-page":"933","DOI":"10.1162\/tacl_a_00681","article-title":"State of what art? A call for multi-prompt llm evaluation","volume":"12","author":"Mizrahi","year":"2024","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.eswa.2026.132645_bib0022","article-title":"Tabular two-dimensional correlation analysis for multifaceted characterization data","author":"Muroga","year":"2024","journal-title":"Applied Spectroscopy"},{"key":"10.1016\/j.eswa.2026.132645_bib0023","series-title":"Synthetic data for deep learning","volume":"vol. 174","author":"Nikolenko","year":"2021"},{"key":"10.1016\/j.eswa.2026.132645_bib0024","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125012","article-title":"Data generation scheme for photovoltaic power forecasting using wasserstein GAN with gradient penalty combined with autoencoder and regression models","volume":"257","author":"Park","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132645_bib0025","series-title":"Advances in neural information processing systems","article-title":"CatBoost: Unbiased boosting with categorical features","volume":"31","author":"Prokhorenkova","year":"2018"},{"issue":"140","key":"10.1016\/j.eswa.2026.132645_bib0026","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"Journal of Machine Learning Research"},{"key":"#cr-split#-10.1016\/j.eswa.2026.132645_bib0027.1","unstructured":"Regulation, Protection, (2016). Regulation"},{"key":"#cr-split#-10.1016\/j.eswa.2026.132645_bib0027.2","unstructured":"(EU) 2016\/679 of the european parliament and of the council. Regulation (eu), 679(2016), 10-3."},{"key":"10.1016\/j.eswa.2026.132645_bib0028","series-title":"Proceedings of the 2020 Conference on empirical methods in natural language processing (EMNLP)","article-title":"How much knowledge can you pack into the parameters of a language model?","author":"Roberts","year":"2020"},{"key":"10.1016\/j.eswa.2026.132645_bib0029","series-title":"Proceedings of the 41st International conference on machine learning","article-title":"Curated LLM: Synergy of LLMs and data curation for tabular augmentation in low-data regimes","author":"Seedat","year":"2024"},{"key":"10.1016\/j.eswa.2026.132645_bib0030","series-title":"Innovative Data Communication Technologies and Application: Proceedings of ICIDCA 2020","first-page":"267","article-title":"A review on word embedding techniques for text classification","author":"Selva Birunda","year":"2021"},{"key":"10.1016\/j.eswa.2026.132645_bib0031","doi-asserted-by":"crossref","unstructured":"Shaheen, M. Y. (2021). Applications of artificial intelligence (AI) in healthcare: A review. ScienceOpen Preprints.","DOI":"10.14293\/S2199-1006.1.SOR-.PPVRY8K.v1"},{"key":"10.1016\/j.eswa.2026.132645_bib0032","series-title":"Proceedings of the 2018 Conference of the North American chapter of the association for computational linguistics: Human language technologies, volume 2 (short papers)","article-title":"Self-attention with relative position representations","author":"Shaw","year":"2018"},{"key":"10.1016\/j.eswa.2026.132645_bib0033","unstructured":"Solatorio, A. V., & Dupriez, O. (2023). RealTabFormer: Generating realistic relational and tabular data using transformers. arXiv: 2302.02041."},{"key":"10.1016\/j.eswa.2026.132645_bib0034","series-title":"Proceedings of the 17th ACM International conference on web search and data mining","first-page":"645","article-title":"Table meets LLM: Can large language models understand structured table data? A benchmark and empirical study","author":"Sui","year":"2024"},{"key":"10.1016\/j.eswa.2026.132645_bib0035","unstructured":"White, J., Fu, Q., Hays, S., Sandborn, M., Olea, C., Gilbert, H., Elnashar, A., Spencer-Smith, J., & Schmidt, D. C. (2023). A prompt pattern catalog to enhance prompt engineering with chatgpt. arXiv: 2302.11382."},{"key":"10.1016\/j.eswa.2026.132645_bib0036","series-title":"Proceedings of the 2020 Conference on empirical methods in natural language processing: System demonstrations","first-page":"38","article-title":"TransFormers: State-of-the-art natural language processing","author":"Wolf","year":"2020"},{"issue":"14","key":"10.1016\/j.eswa.2026.132645_bib0037","doi-asserted-by":"crossref","first-page":"3866","DOI":"10.1002\/cpe.3745","article-title":"Using spearman\u2019s correlation coefficients for exploratory data analysis on big dataset","volume":"28","author":"Xiao","year":"2016","journal-title":"Concurrency and Computation: Practice and Experience"},{"key":"10.1016\/j.eswa.2026.132645_bib0038","series-title":"Advances in neural information processing systems","article-title":"Modeling tabular data using conditional gan","volume":"32","author":"Xu","year":"2019"},{"key":"10.1016\/j.eswa.2026.132645_bib0039","series-title":"Findings of the association for computational linguistics: ACL 2023","first-page":"4756","article-title":"Unified language representation for question answering over text, tables, and images","author":"Yu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132645_bib0040","series-title":"Proceedings of the AAAI Conference on artificial intelligence","first-page":"16803","article-title":"A learnable discrete-prior fusion autoencoder with contrastive learning for tabular data synthesis","volume":"vol. 38","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132645_bib0041","series-title":"Proceedings of the 2023 Conference on empirical methods in natural language processing","first-page":"14836","article-title":"Generative table pre-training empowers models for tabular prediction","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132645_bib0042","unstructured":"Zhao, Z., Birke, R., & Chen, L. (2023). Tabula: Harnessing language models for tabular data synthesis. arXiv: 2310.12746."},{"key":"10.1016\/j.eswa.2026.132645_bib0043","series-title":"Asian conference on machine learning","first-page":"97","article-title":"CTAB-GAN: Effective table data synthesizing","author":"Zhao","year":"2021"},{"key":"10.1016\/j.eswa.2026.132645_bib0044","doi-asserted-by":"crossref","DOI":"10.3389\/fdata.2023.1296508","article-title":"CTAB-GAN+: Enhancing tabular data synthesis","volume":"6","author":"Zhao","year":"2024","journal-title":"Frontiers in big Data"},{"key":"10.1016\/j.eswa.2026.132645_bib0045","unstructured":"Zhou, X., Zhao, X., & Li, G. (2024). LLM-enhanced data management. arXiv: 2402.02643."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015587?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015587?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T02:27:02Z","timestamp":1783132022000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426015587"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":46,"alternative-id":["S0957417426015587"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132645","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"CATS : An enhanced framework in LLM-based tabular data synthesis by correlation augmentation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132645","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132645"}}