{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:40:02Z","timestamp":1783611602659,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761455","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T23:55:33Z","timestamp":1762559733000},"page":"6833-6836","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Generative Models for Synthetic Data: Transforming Data Mining in the GenAI Era"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-4139-7841","authenticated-orcid":false,"given":"Dawei","family":"Li","sequence":"first","affiliation":[{"name":"Arizona State University, Tempe, AZ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8315-7972","authenticated-orcid":false,"given":"Yue","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, IN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6491-4827","authenticated-orcid":false,"given":"Ming","family":"Li","sequence":"additional","affiliation":[{"name":"University of Maryland, College Park, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5348-0632","authenticated-orcid":false,"given":"Tianyi","family":"Zhou","sequence":"additional","affiliation":[{"name":"University of Maryland, College Park, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3574-5665","authenticated-orcid":false,"given":"Xiangliang","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, IN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3264-7904","authenticated-orcid":false,"given":"Huan","family":"Liu","sequence":"additional","affiliation":[{"name":"Arizona State University, Tempe, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Fill In The Gaps: Model Calibration and Generalization with Synthetic Data. arXiv preprint arXiv:2410.10864","author":"Ba Yang","year":"2024","unstructured":"Yang Ba, Michelle V Mancenido, and Rong Pan. 2024. Fill In The Gaps: Model Calibration and Generalization with Synthetic Data. arXiv preprint arXiv:2410.10864 (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves? arXiv preprint arXiv:2410.21259","author":"Bao Han","year":"2024","unstructured":"Han Bao, Yue Huang, Yanbo Wang, Jiayi Ye, Xiangqi Wang, Xiuying Chen, Yue Zhao, Tianyi Zhou, Mohamed Elhoseiny, and Xiangliang Zhang. 2024. AutoBench-V: Can Large Vision-Language Models Benchmark Themselves? arXiv preprint arXiv:2410.21259 (2024)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2024.100293"},{"key":"e_1_3_2_1_4_1","volume-title":"Generative adversarial nets. Advances in neural information processing systems 27","author":"Goodfellow Ian J","year":"2014","unstructured":"Ian J Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, DavidWarde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative adversarial nets. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_5_1","volume-title":"Synthetic data in AI: Challenges, applications, and ethical implications. arXiv preprint arXiv:2401.01629","author":"Hao Shuang","year":"2024","unstructured":"Shuang Hao, Wenfeng Han, Tao Jiang, Yiping Li, Haonan Wu, Chunlin Zhong, Zhangjun Zhou, and He Tang. 2024. Synthetic data in AI: Challenges, applications, and ethical implications. arXiv preprint arXiv:2401.01629 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems 33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020), 6840--6851."},{"key":"e_1_3_2_1_7_1","volume-title":"The Thirteenth International Conference on Learning Representations.","author":"Huang Yue","year":"2024","unstructured":"Yue Huang, Siyuan Wu, Chujie Gao, Dongping Chen, Qihui Zhang, Yao Wan, Tianyi Zhou, Chaowei Xiao, Jianfeng Gao, Lichao Sun, et al. 2024. Datagen: Unified synthetic dataset generation via large language models. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_8_1","volume-title":"Graph generation with diffusion mixture. arXiv preprint arXiv:2302.03596","author":"Jo Jaehyeong","year":"2023","unstructured":"Jaehyeong Jo, Dongki Kim, and Sung Ju Hwang. 2023. Graph generation with diffusion mixture. arXiv preprint arXiv:2302.03596 (2023)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"e_1_3_2_1_10_1","volume-title":"Evaluating Language Models as Synthetic Data Generators. CoRR abs\/2412.03679","author":"Kim Seungone","year":"2024","unstructured":"Seungone Kim, Juyoung Suk, Xiang Yue, Vijay Viswanathan, Seongyun Lee, Yizhong Wang, Kiril Gashteovski, Carolin Lawrence, Sean Welleck, and Graham Neubig. 2024. Evaluating Language Models as Synthetic Data Generators. CoRR abs\/2412.03679 (2024). arXiv:2412.03679 preprint."},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Machine Learning. PMLR, 17564--17579","author":"Kotelnikov Akim","year":"2023","unstructured":"Akim Kotelnikov, Dmitry Baranchuk, Ivan Rubachev, and Artem Babenko. 2023. Tabddpm: Modelling tabular data with diffusion models. In International Conference on Machine Learning. PMLR, 17564--17579."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.119"},{"key":"e_1_3_2_1_13_1","first-page":"1","article-title":"Diffurec: A diffusion model for sequential recommendation","volume":"42","author":"Li Zihao","year":"2023","unstructured":"Zihao Li, Aixin Sun, and Chenliang Li. 2023. Diffurec: A diffusion model for sequential recommendation. ACM Transactions on Information Systems 42, 3 (2023), 1--28.","journal-title":"ACM Transactions on Information Systems"},{"key":"e_1_3_2_1_14_1","volume-title":"CTSyn: A Foundational Model for Cross Tabular Data Generation. arXiv preprint arXiv:2406.04619","author":"Lin Xiaofeng","year":"2024","unstructured":"Xiaofeng Lin, Chenheng Xu, Matthew Yang, and Guang Cheng. 2024. CTSyn: A Foundational Model for Cross Tabular Data Generation. arXiv preprint arXiv:2406.04619 (2024)."},{"key":"e_1_3_2_1_15_1","unstructured":"Chenxi Liu Yongqiang Chen Tongliang Liu Mingming Gong James Cheng Bo Han and Kun Zhang. 2024. Discovery of the HiddenWorld with Large Language Models. (2024). arXiv:2402.03941 [cs.LG] https:\/\/arxiv.org\/abs\/2402.03941"},{"key":"e_1_3_2_1_16_1","unstructured":"Chengyi Liu Wenqi Fan Yunqing Liu Jiatong Li Hang Li Hui Liu Jiliang Tang and Qing Li. [n.d.]. Generative Diffusion Models on Graphs: Methods and Applications. ([n.d.])."},{"key":"e_1_3_2_1_17_1","volume-title":"Empowering Time Series Analysis with Synthetic Data: A Survey and Outlook in the Era of Foundation Models. arXiv preprint arXiv:2503.11411","author":"Liu Xu","year":"2025","unstructured":"Xu Liu, Taha Aksu, Juncheng Liu, Qingsong Wen, Yuxuan Liang, Caiming Xiong, Silvio Savarese, Doyen Sahoo, Junnan Li, and Chenghao Liu. 2025. Empowering Time Series Analysis with Synthetic Data: A Survey and Outlook in the Era of Foundation Models. arXiv preprint arXiv:2503.11411 (2025)."},{"key":"e_1_3_2_1_18_1","volume-title":"Efficacy of Synthetic Data as a Benchmark. CoRR abs\/2409.11968","author":"Maheshwari Gaurav","year":"2024","unstructured":"Gaurav Maheshwari, Dmitry Ivanov, and Kevin El Haddad. 2024. Efficacy of Synthetic Data as a Benchmark. CoRR abs\/2409.11968 (2024). arXiv:2409.11968 preprint."},{"key":"e_1_3_2_1_19_1","volume-title":"Synthetic data generation using large language models: Advances in text and code. arXiv preprint arXiv:2503.14023","author":"Nadas Mihai","year":"2025","unstructured":"Mihai Nadas, Laura Diosan, and Andreea Tomescu. 2025. Synthetic data generation using large language models: Advances in text and code. arXiv preprint arXiv:2503.14023 (2025)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591500"},{"key":"e_1_3_2_1_21_1","first-page":"185","article-title":"Synthetic Data Generation for Fraud Detection Using Diffusion Models","volume":"55","author":"Pushkarenko Yurii","year":"2024","unstructured":"Yurii Pushkarenko and Volodymyr Zaslavskyi. 2024. Synthetic Data Generation for Fraud Detection Using Diffusion Models. Information & Security 55, 2 (2024), 185--198.","journal-title":"Information & Security"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_23_1","volume-title":"The Thirteenth International Conference on Learning Representations.","author":"Shi Juntong","year":"2025","unstructured":"Juntong Shi, Minkai Xu, Harper Hua, Hengrui Zhang, Stefano Ermon, and Jure Leskovec. 2025. TabDiff: a Mixed-type Diffusion Model for Tabular Data Generation. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-024-07566-y"},{"key":"e_1_3_2_1_25_1","volume-title":"Large language models for data annotation and synthesis: A survey. arXiv preprint arXiv:2402.13446","author":"Tan Zhen","year":"2024","unstructured":"Zhen Tan, Dawei Li, Song Wang, Alimohammad Beigi, Bohan Jiang, Amrita Bhattacharjee, Mansooreh Karami, Jundong Li, Lu Cheng, and Huan Liu. 2024. Large language models for data annotation and synthesis: A survey. arXiv preprint arXiv:2402.13446 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Synthesize highdimensional longitudinal electronic health records via hierarchical autoregressive language model. Nature communications 14, 1","author":"Theodorou Brandon","year":"2023","unstructured":"Brandon Theodorou, Cao Xiao, and Jimeng Sun. 2023. Synthesize highdimensional longitudinal electronic health records via hierarchical autoregressive language model. Nature communications 14, 1 (2023), 5305."},{"key":"e_1_3_2_1_27_1","first-page":"48382","article-title":"Stablerep: Synthetic images from text-to-image models make strong visual representation learners","volume":"36","author":"Tian Yonglong","year":"2023","unstructured":"Yonglong Tian, Lijie Fan, Phillip Isola, Huiwen Chang, and Dilip Krishnan. 2023. Stablerep: Synthetic images from text-to-image models make strong visual representation learners. Advances in Neural Information Processing Systems 36 (2023), 48382--48402.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_28_1","volume-title":"Diffusion on Graph: Augmentation of Graph Structure for Node Classification. arXiv preprint arXiv:2503.12563","author":"Wang Yancheng","year":"2025","unstructured":"Yancheng Wang, Changyu Liu, and Yingzhen Yang. 2025. Diffusion on Graph: Augmentation of Graph Structure for Node Classification. arXiv preprint arXiv:2503.12563 (2025)."},{"key":"e_1_3_2_1_29_1","volume-title":"Magpie: Alignment data synthesis from scratch by prompting aligned llms with nothing. arXiv preprint arXiv:2406.08464","author":"Xu Zhangchen","year":"2024","unstructured":"Zhangchen Xu, Fengqing Jiang, Luyao Niu, Yuntian Deng, Radha Poovendran, Yejin Choi, and Bill Yuchen Lin. 2024. Magpie: Alignment data synthesis from scratch by prompting aligned llms with nothing. arXiv preprint arXiv:2406.08464 (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Yang Yao Xin Wang Yijian Qin Zeyang Zhang Wenwu Zhu and Hong Mei. Text-to-graph Generation with Conditional Diffusion Models Guided by Graph-aligned LLMs. ([n.d.])."},{"key":"e_1_3_2_1_31_1","unstructured":"Jiayi Ye Yanbo Wang Yue Huang Dongping Chen Qihui Zhang Nuno Moniz Tian Gao Werner Geyer Chao Huang Pin-Yu Chen et al. 2024. Justice or prejudice? quantifying biases in llm-as-a-judge. arXiv preprint arXiv:2410.02736 (2024)."},{"key":"e_1_3_2_1_32_1","first-page":"55734","article-title":"Large language model as attributed training data generator: A tale of diversity and bias","volume":"36","author":"Yu Yue","year":"2023","unstructured":"Yue Yu, Yuchen Zhuang, Jieyu Zhang, Yu Meng, Alexander J Ratner, Ranjay Krishna, Jiaming Shen, and Chao Zhang. 2023. Large language model as attributed training data generator: A tale of diversity and bias. Advances in Neural Information Processing Systems 36 (2023), 55734--55784.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_33_1","volume-title":"Task Me Anything. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Zhang Jieyu","unstructured":"Jieyu Zhang, Weikai Huang, Zixian Ma, Oscar Michel, Dong He, Tanmay Gupta, Wei-Chiu Ma, Ali Farhadi, Aniruddha Kembhavi, and Ranjay Krishna. [n.d.]. Task Me Anything. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.872"},{"key":"e_1_3_2_1_35_1","volume-title":"Diyi Yang, and Xing Xie.","author":"Zhu Kaijie","year":"2023","unstructured":"Kaijie Zhu, Jiaao Chen, JindongWang, Neil Zhenqiang Gong, Diyi Yang, and Xing Xie. 2023. Dyval: Dynamic evaluation of large language models for reasoning tasks. arXiv preprint arXiv:2309.17167 (2023)."},{"key":"e_1_3_2_1_36_1","volume-title":"Dyval 2: Dynamic evaluation of large language models by meta probing agents. arXiv preprint arXiv:2402.14865","author":"Zhu Kaijie","year":"2024","unstructured":"Kaijie Zhu, Jindong Wang, Qinlin Zhao, Ruochen Xu, and Xing Xie. 2024. Dyval 2: Dynamic evaluation of large language models by meta probing agents. arXiv preprint arXiv:2402.14865 (2024)."}],"event":{"name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","location":"Seoul Republic of Korea","acronym":"CIKM '25","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761455","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:00:21Z","timestamp":1765497621000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761455"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":36,"alternative-id":["10.1145\/3746252.3761455","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761455","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}