{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T18:45:35Z","timestamp":1771267535086,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","funder":[{"name":"DEVCOM Army Research Laboratory","award":["W911NF-24-2-0175"],"award-info":[{"award-number":["W911NF-24-2-0175"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,22]]},"DOI":"10.1145\/3773966.3779380","type":"proceedings-article","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:50:01Z","timestamp":1771264201000},"page":"1160-1164","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["On Causal and Anticausal LLM-based Data Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8552-2681","authenticated-orcid":false,"given":"Bohan","family":"Jiang","sequence":"first","affiliation":[{"name":"Arizona State University, Tempe, AZ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2380-3796","authenticated-orcid":false,"given":"Pingchuan","family":"Ma","sequence":"additional","affiliation":[{"name":"Arizona State University, Tempe, AZ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7936-3931","authenticated-orcid":false,"given":"Zhuoyu","family":"Shi","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0247-4328","authenticated-orcid":false,"given":"Fred","family":"Morstatter","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5489-1633","authenticated-orcid":false,"given":"Adrienne","family":"Raglin","sequence":"additional","affiliation":[{"name":"DEVCOM Army Research Lab, Adelphi, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3264-7904","authenticated-orcid":false,"given":"Huan","family":"Liu","sequence":"additional","affiliation":[{"name":"Arizona State University, Tempe, AZ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","volume-title":"James Glass, and Preslav Nakov.","author":"Baly Ramy","year":"2020","unstructured":"Ramy Baly, Giovanni Da San Martino, James Glass, and Preslav Nakov. 2020. We can detect your bias: Predicting the political ideology of news articles. arXiv preprint arXiv:2010.05338 (2020)."},{"key":"e_1_3_2_1_3_1","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171-4186."},{"key":"e_1_3_2_1_4_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv e-prints (2024) arXiv-2407."},{"key":"e_1_3_2_1_5_1","volume-title":"A kernel two-sample test. The journal of machine learning research","author":"Gretton Arthur","year":"2012","unstructured":"Arthur Gretton, Karsten M Borgwardt, Malte J Rasch, Bernhard Sch\u00f6lkopf, and Alexander Smola. 2012. A kernel two-sample test. The journal of machine learning research, Vol. 13, 1 (2012), 723-773."},{"key":"e_1_3_2_1_6_1","volume-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","unstructured":"D. Janzing and B. Scholkopf. 2015. Semi-supervised interpolation in an anticausal learning scenario. In Journal of machine learning research. doi:10.5555\/2789272.2886811","DOI":"10.5555\/2789272.2886811"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing. 9499-9513","author":"Jin Zhijing","year":"2021","unstructured":"Zhijing Jin, Julius von K\u00fcgelgen, Jingwei Ni, Tejas Vaidhya, Ayush Kaushal, Mrinmaya Sachan, and Bernhard Schoelkopf. 2021. Causal Direction of Data Collection Matters: Implications of Causal and Anticausal Learning for NLP. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing. 9499-9513."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing. 2757-2791","author":"Li Dawei","year":"2025","unstructured":"Dawei Li, Bohan Jiang, Liangjie Huang, Alimohammad Beigi, Chengshuai Zhao, Zhen Tan, Amrita Bhattacharjee, Yuxuan Jiang, Canyu Chen, Tianhao Wu, et al., 2025. From generation to judgment: Opportunities and challenges of llm-as-a-judge. In Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing. 2757-2791."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/2002472.2002491"},{"key":"e_1_3_2_1_11_1","unstructured":"OpenAI. 2025. Openai o3 and o4-mini system card. Technical Report. OpenAI. https:\/\/openai.com\/index\/o3-o4-mini-system-card\/"},{"key":"e_1_3_2_1_12_1","volume-title":"Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084","author":"Reimers Nils","year":"2019","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084 (2019)."},{"key":"e_1_3_2_1_13_1","volume-title":"Conference on Empirical Methods in Natural Language Processing.","author":"Ren Xuan","year":"2024","unstructured":"Xuan Ren, Biao Wu, and Lingqiao Liu. 2024. I Learn Better If You Speak My Language: Understanding the Superior Performance of Fine-Tuning Large Language Models with LLM-Generated Responses. In Conference on Empirical Methods in Natural Language Processing."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.5555\/3042573.3042635"},{"key":"e_1_3_2_1_15_1","volume-title":"Empirical Inference: Festschrift in Honor of Vladimir N. Vapnik","author":"Sch\u00f6lkopf Bernhard","unstructured":"Bernhard Sch\u00f6lkopf, Dominik Janzing, Jonas Peters, Eleni Sgouritsa, Kun Zhang, and Joris Mooij. 2013. Semi-supervised learning in causal and anticausal settings. In Empirical Inference: Festschrift in Honor of Vladimir N. Vapnik. Springer, 129-141."},{"key":"e_1_3_2_1_16_1","first-page":"1422","volume-title":"Proceedings of the International AAAI Conference on Web and Social Media","volume":"18","author":"Shi Zhuoyu","year":"2024","unstructured":"Zhuoyu Shi and Fred Morstatter. 2024. The Diffusion of Causal Language in Social Networks. In Proceedings of the International AAAI Conference on Web and Social Media, Vol. 18. 1422-1435."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.54"},{"key":"e_1_3_2_1_18_1","volume-title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:254877310","author":"Wang Yizhong","year":"2022","unstructured":"Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, Alisa Liu, Noah A. Smith, Daniel Khashabi, and Hannaneh Hajishirzi. 2022. Self-Instruct: Aligning Language Models with Self-Generated Instructions. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:254877310"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3721041.3721046"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_21_1","volume-title":"Large Language Model as Attributed Training Data Generator: A Tale of Diversity and Bias. ArXiv","author":"Yu Yue","year":"2023","unstructured":"Yue Yu, Yuchen Zhuang, Jieyu Zhang, Yu Meng, Alexander J. Ratner, Ranjay Krishna, Jiaming Shen, and Chao Zhang. 2023. Large Language Model as Attributed Training Data Generator: A Tale of Diversity and Bias. ArXiv, Vol. abs\/2306.15895 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the ACM on Web Conference 2025. 2032","author":"Zhang Yunyi","year":"2025","unstructured":"Yunyi Zhang, Ruozhen Yang, Xueqiang Xu, Rui Li, Jinfeng Xiao, Jiaming Shen, and Jiawei Han. 2025. Teleclass: Taxonomy enrichment and llm-enhanced hierarchical text classification with minimal supervision. In Proceedings of the ACM on Web Conference 2025. 2032-2042. endthebibl"}],"event":{"name":"WSDM '26:The Nineteenth ACM International Conference on Web Search and Data Mining","location":"Boise ID USA","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T17:58:29Z","timestamp":1771264709000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3773966.3779380"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,21]]},"references-count":22,"alternative-id":["10.1145\/3773966.3779380","10.1145\/3773966"],"URL":"https:\/\/doi.org\/10.1145\/3773966.3779380","relation":{},"subject":[],"published":{"date-parts":[[2026,2,21]]},"assertion":[{"value":"2026-02-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}