{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:19:42Z","timestamp":1783153182466,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792520","type":"proceedings-article","created":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T21:54:39Z","timestamp":1775771679000},"page":"3240-3250","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Distribution-Aligned Synthetic Text Generation via Tail-Aware Enhancement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-3712-8304","authenticated-orcid":false,"given":"Yuan","family":"Fan","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2625-3896","authenticated-orcid":false,"given":"Xiaoyuan","family":"Liu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China and DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7395-4513","authenticated-orcid":false,"given":"Bo","family":"Liu","sequence":"additional","affiliation":[{"name":"DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5954-8954","authenticated-orcid":false,"given":"Wubing","family":"Wang","sequence":"additional","affiliation":[{"name":"DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4235-2913","authenticated-orcid":false,"given":"Jia","family":"Sun","sequence":"additional","affiliation":[{"name":"DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1674-4701","authenticated-orcid":false,"given":"Wenzhi","family":"Chen","sequence":"additional","affiliation":[{"name":"computer science and Technology, Zhejiang University, hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0947-563X","authenticated-orcid":false,"given":"Huaikang","family":"Fang","sequence":"additional","affiliation":[{"name":"DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5281-8513","authenticated-orcid":false,"given":"Lifeng","family":"Tao","sequence":"additional","affiliation":[{"name":"DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2597-6004","authenticated-orcid":false,"given":"Fan","family":"Mo","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China and DBAPPSecurity Co., Ltd., Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"In-context learning with long-context models: An in-depth exploration. arXiv preprint arXiv:2405.00200","author":"Bertsch Amanda","year":"2024","unstructured":"Amanda Bertsch, Maor Ivgi, Emily Xiao, Uri Alon, Jonathan Berant, Matthew R Gormley, and Graham Neubig. 2024. In-context learning with long-context models: An in-depth exploration. arXiv preprint arXiv:2405.00200 (2024)."},{"key":"e_1_3_2_1_3_1","first-page":"1877","article-title":"Language Models are Few-Shot Learners","volume":"33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. Advances in Neural Information Processing Systems, Vol. 33 (2020), 1877-1901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP46214.2022.9833649"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning. 1772-1783","author":"Chen Yiran","year":"2021","unstructured":"Yiran Chen, Xinyuan Li, Faizan Rehman, Kevin Gimpel, and Kyunghyun Cho. 2021. On Training Instance Selection for Few-Shot Neural Text Generation. In Proceedings of the 38th International Conference on Machine Learning. 1772-1783."},{"key":"e_1_3_2_1_6_1","volume-title":"Model collapse demystified: The case of regression. arXiv preprint arXiv:2402.07712","author":"Dohmatob Elvis","year":"2024","unstructured":"Elvis Dohmatob, Yunzhen Feng, and Julia Kempe. 2024a. Model collapse demystified: The case of regression. arXiv preprint arXiv:2402.07712 (2024)."},{"key":"e_1_3_2_1_7_1","volume-title":"Strong model collapse. arXiv preprint arXiv:2410.04840","author":"Dohmatob Elvis","year":"2024","unstructured":"Elvis Dohmatob, Yunzhen Feng, Arjun Subramonian, and Julia Kempe. 2024b. Strong model collapse. arXiv preprint arXiv:2410.04840 (2024)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","first-page":"76852","DOI":"10.52202\/075280-3358","article-title":"Flocks of stochastic parrots: Differentially private prompt learning for large language models","volume":"36","author":"Duan Haonan","year":"2023","unstructured":"Haonan Duan, Adam Dziedzic, Nicolas Papernot, and Franziska Boenisch. 2023. Flocks of stochastic parrots: Differentially private prompt learning for large language models. Advances in Neural Information Processing Systems, Vol. 36 (2023), 76852-76871.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-79228-4_1"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/11681878_14"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Cynthia Dwork Aaron Roth et al. 2014. The algorithmic foundations of differential privacy. Foundations and trends\u00ae in theoretical computer science Vol. 9 3-4 (2014) 211-407.","DOI":"10.1561\/0400000042"},{"key":"e_1_3_2_1_12_1","volume-title":"ICLR Workshop on Navigating and Addressing Data Problems for Foundation \u2026.","author":"Feng Yunzhen","year":"2024","unstructured":"Yunzhen Feng, Elvis Dohmatob, Pu Yang, Francois Charton, and Julia Kempe. 2024. A tale of tails: Model collapse as a change of scaling laws. ICLR Workshop on Navigating and Addressing Data Problems for Foundation \u2026."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/3241691.3241693"},{"key":"e_1_3_2_1_14_1","unstructured":"Matthias Gerstgrasser Rylan Schaeffer Apratim Dey Rafael Rafailov Henry Sleight John Hughes Tomasz Korbak Rajashree Agrawal Dhruv Pai Andrey Gromov et al. 2024. Is model collapse inevitable? breaking the curse of recursion by accumulating real and synthetic data. arXiv preprint arXiv:2404.01413 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Harnessing the power of synthetic data in healthcare: innovation, application, and privacy. NPJ digital medicine","author":"Giuffr\u00e8 Mauro","year":"2023","unstructured":"Mauro Giuffr\u00e8 and Dennis L Shung. 2023. Harnessing the power of synthetic data in healthcare: innovation, application, and privacy. NPJ digital medicine, Vol. 6, 1 (2023), 186."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.647"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1017\/pan.2024.31"},{"key":"e_1_3_2_1_18_1","volume-title":"Pre-text: Training language models on private federated data in the age of llms. arXiv preprint arXiv:2406.02958","author":"Hou Charlie","year":"2024","unstructured":"Charlie Hou, Akshat Shrivastava, Hongyuan Zhan, Rylan Conway, Trang Le, Adithya Sagar, Giulia Fanti, and Daniel Lazar. 2024. Pre-text: Training language models on private federated data in the age of llms. arXiv preprint arXiv:2406.02958 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP54263.2024.00002"},{"key":"e_1_3_2_1_20_1","unstructured":"Y Inc. 2023. Yelp dataset. https:\/\/www.yelp.com\/dataset\/."},{"key":"e_1_3_2_1_21_1","volume-title":"Enhancing Spatial Reasoning in Vision-Language Models via Chain-of-Thought Prompting and Reinforcement Learning. arXiv preprint arXiv:2507.13362","author":"Ji Binbin","year":"2025","unstructured":"Binbin Ji, Siddharth Agrawal, Qiance Tang, and Yvonne Wu. 2025. Enhancing Spatial Reasoning in Vision-Language Models via Chain-of-Thought Prompting and Reinforcement Learning. arXiv preprint arXiv:2507.13362 (2025)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3362821"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/S14-2003"},{"key":"e_1_3_2_1_24_1","volume-title":"Decoupling Representation and Classifier for Long-Tailed Recognition. In International Conference on Learning Representations.","author":"Kang Bingyi","year":"2020","unstructured":"Bingyi Kang, Saining Xie, Marcus Rohrbach, Zhicheng Yan, Albert Gordo, Jiashi Feng, and Yannis Kalantidis. 2020. Decoupling Representation and Classifier for Long-Tailed Recognition. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00498"},{"key":"e_1_3_2_1_26_1","volume-title":"Federated domain-specific knowledge transfer on large language models using synthetic data. arXiv preprint arXiv:2405.14212","author":"Li Haoran","year":"2024","unstructured":"Haoran Li, Xinyuan Zhao, Dadi Guo, Hanlin Gu, Ziqian Zeng, Yuxing Han, Yangqiu Song, Lixin Fan, and Qiang Yang. 2024. Federated domain-specific knowledge transfer on large language models using synthetic data. arXiv preprint arXiv:2405.14212 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Differentially private synthetic data via foundation model apis 1: Images. arXiv preprint arXiv:2305.15560","author":"Lin Zinan","year":"2023","unstructured":"Zinan Lin, Sivakanth Gopi, Janardhan Kulkarni, Harsha Nori, and Sergey Yekhanin. 2023. Differentially private synthetic data via foundation model apis 1: Images. arXiv preprint arXiv:2305.15560 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019b. RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR, Vol. abs\/1907.11692 (2019). arXiv:1907.11692 http:\/\/arxiv.org\/abs\/1907.11692"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00264"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems Vol. 35 (2022) 27730-27744.","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_31_1","volume-title":"Training question answering models from synthetic data. arXiv preprint arXiv:2002.09599","author":"Puri Raul","year":"2020","unstructured":"Raul Puri, Ryan Spring, Mostofa Patwary, Mohammad Shoeybi, and Bryan Catanzaro. 2020. Training question answering models from synthetic data. arXiv preprint arXiv:2002.09599 (2020)."},{"key":"e_1_3_2_1_32_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog Vol. 1 8 (2019) 9."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19784-0_25"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00033"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00936"},{"key":"e_1_3_2_1_36_1","volume-title":"Seon Joo Kim, and Jonghyun Choi","author":"Seo Minhyuk","year":"2024","unstructured":"Minhyuk Seo, Seongwon Cho, Minjae Lee, Diganta Misra, Hyeonbeom Choi, Seon Joo Kim, and Jonghyun Choi. 2024. Just say the name: Online continual learning with category names only via data generation. arXiv preprint arXiv:2403.10853 (2024)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1361"},{"key":"e_1_3_2_1_38_1","first-page":"3","article-title":"Membership inference attacks against machine learning models","author":"Shokri Reza","year":"2017","unstructured":"Reza Shokri, Marco Stronati, Congzheng Song, and Vitaly Shmatikov. 2017. Membership inference attacks against machine learning models. In Proceedings of IEEE S&P. 3-18.","journal-title":"Proceedings of IEEE S&P."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-024-07566-y"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2023.3331024"},{"key":"e_1_3_2_1_41_1","volume-title":"A data-efficient strategy for building high-performing medical foundation models. Nature Biomedical Engineering","author":"Sun Yuqi","year":"2025","unstructured":"Yuqi Sun, Weimin Tan, Zhuoyao Gu, Ruian He, Siyuan Chen, Miao Pang, and Bo Yan. 2025. A data-efficient strategy for building high-performing medical foundation models. Nature Biomedical Engineering (2025), 1-13."},{"key":"e_1_3_2_1_42_1","first-page":"476","volume-title":"Nature","volume":"625","author":"Trinh Trieu H","year":"2024","unstructured":"Trieu H Trinh, Yuhuai Wu, Quoc V Le, He He, and Thang Luong. 2024. Solving olympiad geometry without human demonstrations. Nature, Vol. 625, 7995 (2024), 476-482."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3490035.3490300"},{"key":"e_1_3_2_1_44_1","volume-title":"KnowledgeSG: Privacy-preserving synthetic text generation with knowledge distillation from server. arXiv preprint arXiv:2410.05725","author":"Wang Wenhao","year":"2024","unstructured":"Wenhao Wang, Xiaoyu Liang, Rui Ye, Jingyi Chai, Siheng Chen, and Yanfeng Wang. 2024. KnowledgeSG: Privacy-preserving synthetic text generation with knowledge distillation from server. arXiv preprint arXiv:2410.05725 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Self-instruct: Aligning language models with self-generated instructions. arXiv preprint arXiv:2212.10560","author":"Wang Yizhong","year":"2022","unstructured":"Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, Alisa Liu, Noah A Smith, Daniel Khashabi, and Hannaneh Hajishirzi. 2022. Self-instruct: Aligning language models with self-generated instructions. arXiv preprint arXiv:2212.10560 (2022)."},{"key":"e_1_3_2_1_46_1","volume-title":"Yin Tat Lee, et al","author":"Xie Chulin","year":"2024","unstructured":"Chulin Xie, Zinan Lin, Arturs Backurs, Sivakanth Gopi, Da Yu, Huseyin A Inan, Harsha Nori, Haotian Jiang, Huishuai Zhang, Yin Tat Lee, et al., 2024. Differentially private synthetic data via foundation model apis 2: Text. arXiv preprint arXiv:2403.01749 (2024)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.916"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.194"},{"key":"e_1_3_2_1_49_1","volume-title":"Synthetic text generation with differential privacy: A simple and practical recipe. arXiv preprint arXiv:2210.14348","author":"Yue Xiang","year":"2022","unstructured":"Xiang Yue, Huseyin A Inan, Xuechen Li, Girish Kumar, Julia McAnallen, Hoda Shajari, Huan Sun, David Levitan, and Robert Sim. 2022. Synthetic text generation with differential privacy: A simple and practical recipe. arXiv preprint arXiv:2210.14348 (2022)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2025.3539314"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0819"},{"key":"e_1_3_2_1_52_1","volume-title":"How to Synthesize Text Data without Model Collapse? arXiv preprint arXiv:2412.14689","author":"Zhu Xuekai","year":"2024","unstructured":"Xuekai Zhu, Daixuan Cheng, Hengli Li, Kaiyan Zhang, Ermo Hua, Xingtai Lv, Ning Ding, Zhouhan Lin, Zilong Zheng, and Bowen Zhou. 2024. How to Synthesize Text Data without Model Collapse? arXiv preprint arXiv:2412.14689 (2024)."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792520","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:59:46Z","timestamp":1783151986000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792520"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":52,"alternative-id":["10.1145\/3774904.3792520","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792520","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}