{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T04:17:54Z","timestamp":1784607474151,"version":"3.55.0"},"reference-count":161,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T00:00:00Z","timestamp":1765411200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T00:00:00Z","timestamp":1765411200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100014736","name":"Lingnan University","doi-asserted-by":"publisher","award":["Faculty Research Grants (SDS24A8 and SDS24A19) and the Direct Grant (DR25E8)"],"award-info":[{"award-number":["Faculty Research Grants (SDS24A8 and SDS24A19) and the Direct Grant (DR25E8)"]}],"id":[{"id":"10.13039\/501100014736","id-type":"DOI","asserted-by":"publisher"}]},{"name":"2023 Nanjing International\/Hong Kong, Macao, and Taiwan Science and Technology Cooperation Program","award":["202308010"],"award-info":[{"award-number":["202308010"]}]},{"name":"Research Grants Council of the Hong Kong Special Administrative Region, China","award":["R1015-23"],"award-info":[{"award-number":["R1015-23"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-025-11405-5","type":"journal-article","created":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T04:52:14Z","timestamp":1765428734000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["Text data augmentation for large language models: a comprehensive survey of methods, challenges, and opportunities"],"prefix":"10.1007","volume":"59","author":[{"given":"Yaping","family":"Chai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoran","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joe S.","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,11]]},"reference":[{"key":"11405_CR1","doi-asserted-by":"publisher","unstructured":"Ahmad W, Chi J, Tian Y, Chang K-W (2020) PolicyQA: a reading comprehension dataset for privacy policies. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 743\u2013749. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.66","DOI":"10.18653\/v1\/2020.findings-emnlp.66"},{"key":"11405_CR2","doi-asserted-by":"publisher","unstructured":"Anaby-Tavor A, Carmeli B, Goldbraich E, Kantor A, Kour G, Shlomov S, Tepper N, Zwerdling N (2020) Do not have enough data? deep learning to the rescue! In: The Thirty-Fourth AAAI Conference on Artificial Intelligence, AAAI 2020, The Thirty-Second Innovative Applications of Artificial Intelligence Conference, IAAI 2020, The Tenth AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI, pp. 7383\u20137390. AAAI Press, New York, NY, USA. https:\/\/doi.org\/10.1609\/AAAI.V34I05.6233","DOI":"10.1609\/AAAI.V34I05.6233"},{"key":"11405_CR3","unstructured":"Araci D (2019) Finbert: financial sentiment analysis with pre-trained language models. Preprint at arXiv:1908.10063"},{"key":"11405_CR4","doi-asserted-by":"crossref","unstructured":"Baek J, Aji AF, Saffari A (2023) Knowledge-augmented language model prompting for zero-shot knowledge graph question answering. Preprint at arXiv:2306.04136","DOI":"10.18653\/v1\/2023.nlrse-1.7"},{"issue":"7","key":"11405_CR5","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1145\/3544558","volume":"55","author":"M Bayer","year":"2023","unstructured":"Bayer M, Kaufhold M, Reuter C (2023) A survey on data augmentation for text classification. ACM Comput Surv 55(7):146\u2013114639. https:\/\/doi.org\/10.1145\/3544558","journal-title":"ACM Comput Surv"},{"key":"11405_CR6","doi-asserted-by":"crossref","unstructured":"Black S et al (2022) Gpt-neox-20b: An open-source autoregressive language model. Preprint at arXiv:2204.06745","DOI":"10.18653\/v1\/2022.bigscience-1.9"},{"key":"11405_CR7","doi-asserted-by":"crossref","unstructured":"Bonifacio LH, Abonizio HQ, Fadaee M, Nogueira RF (2022) InPars: data augmentation for information retrieval using large language models. Preprint at arXiv:2202.05144","DOI":"10.1145\/3477495.3531863"},{"key":"11405_CR8","doi-asserted-by":"publisher","unstructured":"Cai X, Xiao M, Ning Z, Zhou Y (2023) Resolving the imbalance issue in hierarchical disciplinary topic inference via llm-based data augmentation. In: IEEE International Conference on Data Mining, pp. 956\u2013961. IEEE, Shanghai, China. https:\/\/doi.org\/10.1109\/ICDM58522.2023.00107","DOI":"10.1109\/ICDM58522.2023.00107"},{"key":"11405_CR9","doi-asserted-by":"publisher","unstructured":"Castelli V et al (2020) The TechQA dataset. In: Proceedings of the 58th annual meeting of the association for computational linguistics, pp. 1269\u20131278. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.117","DOI":"10.18653\/v1\/2020.acl-main.117"},{"key":"11405_CR10","unstructured":"Chai Y, Xie H, Qin JS (2025) Semantic-preserved augmentation with confidence-weighted fine-tuning for aspect category sentiment analysis. Preprint at arXiv:2506.07148"},{"key":"11405_CR11","doi-asserted-by":"publisher","first-page":"2851","DOI":"10.7717\/peerj-cs.2851","volume":"11","author":"L Chen","year":"2025","unstructured":"Chen L, Shang S, Wang Y (2025) Bridging resource gaps in cross-lingual sentiment analysis: adaptive self-alignment with data augmentation and transfer learning. PeerJ Comput Sci 11:2851. https:\/\/doi.org\/10.7717\/peerj-cs.2851","journal-title":"PeerJ Comput Sci"},{"key":"11405_CR12","doi-asserted-by":"publisher","unstructured":"Chen G, Chen S, Liu Z, Jiang F, Wang B (2024) Humans or llms as the judge? A study on judgement bias. In: Proceedings of the 2024 conference on empirical methods in natural language processing, EMNLP, pp. 8301\u20138327. Association for Computational Linguistics, Miami, FL, USA. https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.474","DOI":"10.18653\/v1\/2024.emnlp-main.474"},{"key":"11405_CR13","doi-asserted-by":"publisher","unstructured":"Chen Z, Chen J, Singh AK, Sra M (2024) Xplainllm: a knowledge-augmented dataset for reliable grounded explanations in llms. In: Proceedings of the 2024 conference on empirical methods in natural language processing, EMNLP, pp. 7578\u20137596. Association for Computational Linguistics, Miami, FL, USA. https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.432","DOI":"10.18653\/v1\/2024.emnlp-main.432"},{"key":"11405_CR14","doi-asserted-by":"publisher","unstructured":"Chen G, Chen Y, Wang Y, Li VOK (2020) Lexical-constraint-aware neural machine translation via data augmentation. In: Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, IJCAI, pp. 3587\u20133593. ijcai.org, Virtual Event. https:\/\/doi.org\/10.24963\/IJCAI.2020\/496","DOI":"10.24963\/IJCAI.2020\/496"},{"key":"11405_CR15","doi-asserted-by":"publisher","unstructured":"Chen H, Dou Z, Mao K, Liu J, Zhao Z (2024) Generalizing conversational dense retrieval via llm-cognition data augmentation. In: Proceedings of the 62nd annual meeting of the association for computational linguistics (Volume 1: Long Papers), ACL, pp. 2700\u20132718. Association for Computational Linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/V1\/2024.ACL-LONG.149","DOI":"10.18653\/V1\/2024.ACL-LONG.149"},{"key":"11405_CR16","unstructured":"Cheng Q, Huang J, Duan Y (2022) Semantically consistent data augmentation for neural machine translation via conditional masked language model. In: Proc. Int. Conf. Comput. Linguist. (COLING), pp. 5148\u20135157. Int. Comm. Comput. Linguistics, Seoul, South Korea. https:\/\/aclanthology.org\/2022.coling-1.457"},{"key":"11405_CR17","doi-asserted-by":"publisher","unstructured":"Chen X, Xie H, Qin SJ, Chai Y, Tao X, Wang FL (2024) Cognitive-inspired deep learning models for aspect-based sentiment analysis: a retrospective overview and bibliometric analysis. Cogn Comput 1\u201339. https:\/\/doi.org\/10.1007\/S12559-024-10331-Y","DOI":"10.1007\/S12559-024-10331-Y"},{"key":"11405_CR18","unstructured":"Chen B, Zhang Z, Langren\u00e9 N, Zhu S (2023) Unleashing the potential of prompt engineering in large language models: a comprehensive review. Preprint at arXiv:2310.14735"},{"key":"11405_CR19","doi-asserted-by":"publisher","unstructured":"Chowdhury AG, Chadha A (2024) Generative data augmentation using llms improves distributional robustness in question answering. In: Proceedings of the 18th Conference of the European chapter of the association for computational linguistics, pp. 258\u2013265. Association for Computational Linguistics, Malta. https:\/\/doi.org\/10.18653\/v1\/2024.eacl-srw.20","DOI":"10.18653\/v1\/2024.eacl-srw.20"},{"key":"11405_CR20","doi-asserted-by":"crossref","unstructured":"Cicekyurt E, Bakal G (2025) Enhancing sentiment analysis in stock market tweets through bert-based knowledge transfer. Comput Econ 1\u201323","DOI":"10.1007\/s10614-025-10901-8"},{"key":"11405_CR21","doi-asserted-by":"crossref","unstructured":"Coucke A, Saade A, Ball A, Bluche T, Caulier A, Leroy D, Doumouro C, Gisselbrecht T, Caltagirone F, Lavril T, Primet M, Dureau J (2018) Snips Voice Platform: an embedded Spoken Language Understanding system for private-by-design voice interfaces. Preprint at arXiv:1805.10190","DOI":"10.1109\/EMC2-NIPS53020.2019.00021"},{"issue":"3","key":"11405_CR22","doi-asserted-by":"publisher","first-page":"907","DOI":"10.1109\/TBDATA.2025.3536934","volume":"11","author":"H Dai","year":"2025","unstructured":"Dai H et al (2025) Auggpt: leveraging chatgpt for text data augmentation. IEEE Trans Big Data 11(3):907\u2013918. https:\/\/doi.org\/10.1109\/TBDATA.2025.3536934","journal-title":"IEEE Trans Big Data"},{"key":"11405_CR23","unstructured":"Dai Z, Zhao VY, Ma J, Luan Y, Ni J, Lu J, Bakalov A, Guu K, Hall KB, Chang M (2023) Promptagator: few-shot dense retrieval from 8 examples. In: The Eleventh International Conference on Learning Representations. OpenReview.net, Kigali, Rwanda"},{"issue":"2","key":"11405_CR24","doi-asserted-by":"publisher","first-page":"265","DOI":"10.24271\/psr.2024.440793.1484","volume":"6","author":"F Daneshfar","year":"2024","unstructured":"Daneshfar F (2024) Enhancing low-resource sentiment analysis: a transfer learning approach. Passer J Basic Appl Sci 6(2):265\u2013274","journal-title":"Passer J Basic Appl Sci"},{"key":"11405_CR25","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang M, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4171\u20134186. Association for Computational Linguistics, USA. https:\/\/doi.org\/10.18653\/V1\/N19-1423","DOI":"10.18653\/V1\/N19-1423"},{"key":"11405_CR26","doi-asserted-by":"publisher","unstructured":"Ding B, Qin C, Zhao R, Luo T, Li X, Chen G, Xia W, Hu J, Luu AT, Joty S (2024) Data augmentation using llms: Data perspectives, learning paradigms and challenges. In: Findings of the association for computational linguistics, ACL, pp. 1679\u20131705. Association for Computational Linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/v1\/2024.findings-acl.97","DOI":"10.18653\/v1\/2024.findings-acl.97"},{"key":"11405_CR27","doi-asserted-by":"publisher","unstructured":"Du X, Ji H (2022) Retrieval-augmented generative question answering for event argument extraction. In: Proceedings of the 2022 conference on empirical methods in natural language processing, EMNLP, pp. 4649\u20134666. Association for computational linguistics, United Arab Emirates. https:\/\/doi.org\/10.18653\/V1\/2022.EMNLP-MAIN.307","DOI":"10.18653\/V1\/2022.EMNLP-MAIN.307"},{"key":"11405_CR28","unstructured":"Edwards A, Ushio A, Camacho-Collados J, Ribaupierre H, Preece AD (2021) Guiding generative language models for data augmentation in few-shot text classification. Preprint at arXiv:2111.09064"},{"key":"11405_CR29","doi-asserted-by":"crossref","unstructured":"Feng SY, Gangal V, Kang D, Mitamura T, Hovy EH (2020) Genaug: data augmentation for finetuning text generators. Preprint at arXiv:2010.01794","DOI":"10.18653\/v1\/2020.deelio-1.4"},{"key":"11405_CR30","doi-asserted-by":"publisher","unstructured":"Feng SY, Gangal V, Wei J, Chandar S, Vosoughi S, Mitamura T, Hovy EH (2021) A survey of data augmentation approaches for NLP. In: Findings of the Association for Computational Linguistics: ACL\/IJCNLP. Findings of ACL, vol. ACL\/IJCNLP 2021, pp. 968\u2013988. association for computational linguistics, Online. https:\/\/doi.org\/10.18653\/v1\/2021.findings-acl.84","DOI":"10.18653\/v1\/2021.findings-acl.84"},{"key":"11405_CR31","doi-asserted-by":"crossref","unstructured":"Gao M, Hu X, Ruan J, Pu X, Wan X (2025) Llm-based NLG evaluation: current status and challenges. Preprint at arXiv:2402.01383","DOI":"10.1162\/coli_a_00561"},{"key":"11405_CR32","doi-asserted-by":"publisher","unstructured":"Gao J, Pi R, Lin Y, Xu H, Ye J, Wu Z, Zhang W, Liang X, Li Z, Kong L (2023) Self-guided noise-free data generation for efficient zero-shot learning. In: The Eleventh international conference on learning representations, pp. 1\u20135. OpenReview.net, Kigali, Rwanda. https:\/\/doi.org\/10.2139\/ssrn.4545107","DOI":"10.2139\/ssrn.4545107"},{"key":"11405_CR33","unstructured":"Gao Y, Xiong Y, Gao X, Jia K, Pan J, Bi Y, Dai Y, Sun J, Wang M, Wang H (2024) Retrieval-augmented generation for large language models: a survey. Preprint at arXiv:2312.10997"},{"key":"11405_CR34","doi-asserted-by":"publisher","unstructured":"Gao T, Yao X, Chen D (2021) Simcse: Simple contrastive learning of sentence embeddings. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP, pp. 6894\u20136910. Association for computational linguistics, Dominican Republic. https:\/\/doi.org\/10.18653\/V1\/2021.EMNLP-MAIN.552","DOI":"10.18653\/V1\/2021.EMNLP-MAIN.552"},{"key":"11405_CR35","doi-asserted-by":"publisher","unstructured":"Gao T, Yen H, Yu J, Chen D (2023) Enabling large language models to generate text with citations. In: Proceedings of the 2023 Conference on empirical methods in natural language processing, pp. 6465\u20136488. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.398","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.398"},{"key":"11405_CR36","unstructured":"Gonzalez RA, DiPaola S (2024) Exploring augmentation and cognitive strategies for AI based synthetic personae. Preprint at arXiv:2404.10890"},{"key":"11405_CR37","unstructured":"Guo B, Gong Y, Shen Y, Han S, Huang H, Duan N, Chen W (2022) GENIUS: Sketch-based language model pre-training via extreme and selective masking for text generation and augmentation. Preprint at arXiv:2211.10330"},{"key":"11405_CR38","doi-asserted-by":"publisher","unstructured":"Gupta P, Bigham JP, Tsvetkov Y, Pavel A (2021) Controlling dialogue generation with semantic exemplars. In: Proceedings of the 2021 conference of the north American chapter of the association for computational linguistics: human language technologies, NAACL-HLT, pp. 3018\u20133029. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2021.NAACL-MAIN.240","DOI":"10.18653\/V1\/2021.NAACL-MAIN.240"},{"key":"11405_CR39","unstructured":"Hinton GE, Vinyals O, Dean J (2015) Distilling the knowledge in a neural network. Preprint at arXiv:1503.02531"},{"key":"11405_CR40","doi-asserted-by":"publisher","unstructured":"Hofst\u00e4tter S, Lin S, Yang J, Lin J, Hanbury A (2021) Efficiently teaching an effective dense retriever with balanced topic aware sampling. In: SIGIR \u201921: The 44th International ACM SIGIR conference on research and development in information retrieval, pp. 113\u2013122. ACM, Canada. https:\/\/doi.org\/10.1145\/3404835.3462891","DOI":"10.1145\/3404835.3462891"},{"key":"11405_CR41","doi-asserted-by":"publisher","unstructured":"Honovich O, Scialom T, Levy O, Schick T (2023) Unnatural instructions: Tuning language models with (almost) no human labor. In: Proceedings of the 61st Annual meeting of the association for computational linguistics (Volume 1: Long Papers), pp. 14409\u201314428. Association for computational linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.806","DOI":"10.18653\/v1\/2023.acl-long.806"},{"key":"11405_CR42","doi-asserted-by":"publisher","unstructured":"Hsu T, Chen C, Huang H, Chen H (2021) Semantics-preserved data augmentation for aspect-based sentiment analysis. In: Proc. Conf. Empirical Methods Nat. Lang. Process. (EMNLP), pp. 4417\u20134422. Assoc. Comput. Linguistics, Punta Cana, Dominican Republic. https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.362","DOI":"10.18653\/v1\/2021.emnlp-main.362"},{"key":"11405_CR43","doi-asserted-by":"publisher","unstructured":"Huang Q, Fu S, Liu X, Wang W, Ko T, Zhang Y, Tang L (2023) Learning retrieval augmentation for personalized dialogue generation. In: Proceedings of the 2023 Conference on empirical methods in natural language processing, EMNLP, pp. 2523\u20132540. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.154","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.154"},{"key":"11405_CR44","doi-asserted-by":"publisher","unstructured":"Huang K, Hsu I, Natarajan P, Chang K, Peng N (2022) Multilingual generative language models for zero-shot cross-lingual event argument extraction. In: Proceedings of the 60th annual meeting of the association for computational linguistics (Volume 1: Long Papers), ACL, pp. 4633\u20134646. Association for Computational Linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/V1\/2022.ACL-LONG.317","DOI":"10.18653\/V1\/2022.ACL-LONG.317"},{"key":"11405_CR45","unstructured":"Huang W, Qin H, Liu Y, Li Y, Liu X, Benini L, Magno M, Qi X (2024) SliM-LLM: salience-driven mixed-precision quantization for large language models. Preprint at arXiv:2405.14917"},{"key":"11405_CR46","unstructured":"Humeau S, Shuster K, Lachaux M-A, Weston J (2019) Poly-encoders: architectures and pre-training strategies for fast and accurate multi-sentence scoring. In: International conference on learning representations. https:\/\/openreview.net\/forum?id=SkxgnnNFvH"},{"key":"11405_CR47","unstructured":"Hu T, Zhou X (2024) Unveiling LLM evaluation focused on metrics: Challenges and solutions. Preprint at arXiv:2404.09135"},{"key":"11405_CR48","unstructured":"Izacard G, Caron M, Hosseini L, Riedel S, Bojanowski P, Joulin A, Grave E (2022) Unsupervised dense information retrieval with contrastive learning. Trans. Mach. Learn. Res. https:\/\/openreview.net\/forum?id=jKN1pXi7b0"},{"key":"11405_CR49","unstructured":"Jo B, Heo T, Park Y, Yoo Y, Cho W, Kim K (2022) DAGAM: data augmentation with generation and modification. Preprint at arXiv:2204.02633"},{"key":"11405_CR50","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1162\/TACL_A_00300","volume":"8","author":"M Joshi","year":"2020","unstructured":"Joshi M, Chen D, Liu Y, Weld DS, Zettlemoyer L, Levy O (2020) Spanbert: Improving pre-training by representing and predicting spans. Trans Assoc Comput Linguistics 8:64\u201377. https:\/\/doi.org\/10.1162\/TACL_A_00300","journal-title":"Trans Assoc Comput Linguistics"},{"key":"11405_CR51","unstructured":"Kaddour J, Liu Q (2024) Synthetic data generation in low-resource settings via fine-tuning of large language models. Preprint at arXiv:2310.01119"},{"key":"11405_CR52","doi-asserted-by":"publisher","unstructured":"Karpukhin V, Oguz B, Min S, Lewis PSH, Wu L, Edunov S, Chen D, Yih W (2020) Dense passage retrieval for open-domain question answering. In: Proceedings of the 2020 conference on empirical methods in natural language processing, EMNLP. Association for computational linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2020.EMNLP-MAIN.550","DOI":"10.18653\/V1\/2020.EMNLP-MAIN.550"},{"key":"11405_CR53","unstructured":"Ko J, Chen T, Kim S, Ding T, Liang L, Zharkov I, Yun S (2025) Distillm-2: A contrastive approach boosts the distillation of llms. Preprint at arXiv:2503.07067"},{"key":"11405_CR54","doi-asserted-by":"publisher","unstructured":"Komeili M, Shuster K, Weston J (2022) Internet-augmented dialogue generation. In: Proceedings of the 60th annual meeting of the association for computational linguistics (Volume 1: Long Papers), ACL, pp. 8460\u20138478. Association for computational linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/V1\/2022.ACL-LONG.579","DOI":"10.18653\/V1\/2022.ACL-LONG.579"},{"key":"11405_CR55","doi-asserted-by":"crossref","unstructured":"Kulh\u00e1nek J, Hude\u010dek V, Nekvinda T, Du\u0161ek O (2021) AuGPT: Auxiliary tasks and data augmentation for end-to-end dialogue with pre-trained language models. Preprint at arXiv:2102.05126","DOI":"10.18653\/v1\/2021.nlp4convai-1.19"},{"key":"11405_CR56","doi-asserted-by":"crossref","unstructured":"Kumar V, Choudhary A, Cho E (2020) Data augmentation using pre-trained transformer models. Preprint at arXiv:2003.02245","DOI":"10.18653\/v1\/2020.lifelongnlp-1.3"},{"key":"11405_CR57","doi-asserted-by":"publisher","unstructured":"Lai VD, Ngo NT, Veyseh APB, Man H, Dernoncourt F, Bui T, Nguyen TH (2023) Chatgpt beyond english: Towards a comprehensive evaluation of large language models in multilingual learning. In: Findings of the association for computational linguistics: EMNLP, pp. 13171\u201313189. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.FINDINGS-EMNLP.878","DOI":"10.18653\/V1\/2023.FINDINGS-EMNLP.878"},{"key":"11405_CR58","unstructured":"Lazaridou A, Gribovskaya E, Stokowiec W, Grigorev N (2022) Internet-augmented language models through few-shot prompting for open-domain question answering. Preprint at arXiv:2203.05115"},{"key":"11405_CR59","doi-asserted-by":"publisher","unstructured":"Lee N, Wattanawong T, Kim S, Mangalam K, Shen S, Anumanchipalli G, Mahoney MW, Keutzer K, Gholami A (2024) LLM2LLM: boosting llms with novel iterative data enhancement. In: Findings of the association for computational linguistics, pp. 6498\u20136526. Association for computational linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/V1\/2024.FINDINGS-ACL.388","DOI":"10.18653\/V1\/2024.FINDINGS-ACL.388"},{"key":"11405_CR60","unstructured":"Lewis PSH, Perez E, Piktus A, Petroni F, Karpukhin V, Goyal N, K\u00fcttler H, Lewis M, Yih W, Rockt\u00e4schel T, Riedel S, Kiela D (2020) Retrieval-augmented generation for knowledge-intensive NLP tasks. In: Advances in neural information processing systems 33: Annual conference on neural information processing systems 2020, NeurIPS 2020, Virtual Event. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/6b493230205f780e1bc26945df7481e5-Abstract.html"},{"key":"11405_CR61","doi-asserted-by":"publisher","unstructured":"Lewis M, Liu Y, Goyal N, Ghazvininejad M, Mohamed A, Levy O, Stoyanov V, Zettlemoyer L (2020) BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. In: Proceedings of the 58th annual meeting of the association for computational linguistics, pp. 7871\u20137880. Association for computational linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2020.ACL-MAIN.703","DOI":"10.18653\/V1\/2020.ACL-MAIN.703"},{"key":"11405_CR62","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1016\/j.aiopen.2022.03.001","volume":"3","author":"B Li","year":"2022","unstructured":"Li B, Hou Y, Che W (2022) Data augmentation approaches in natural language processing: A survey. AI Open 3:71\u201390. https:\/\/doi.org\/10.1016\/j.aiopen.2022.03.001","journal-title":"AI Open"},{"key":"11405_CR63","unstructured":"Liang P et al (2023) Holistic evaluation of language models. Trans Mach Learn Res https:\/\/openreview.net\/forum?id=iO4LZibEqW"},{"key":"11405_CR64","unstructured":"Li M, Chen H, Wang Y, Zhu T, Zhang W, Zhu K, Wong K, Wang J (2025) Understanding and mitigating the bias inheritance in LLM-based data augmentation on downstream tasks. Preprint at arXiv:2502.04419"},{"key":"11405_CR65","unstructured":"Li D, Li Y, Mekala D, Li S, Wang Y, Wang X, Hogan W, Shang J (2023) DAIL: data augmentation for in-context learning via self-paraphrase. Preprint at arXiv:2311.03319"},{"issue":"1","key":"11405_CR66","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1002\/ARIS.1440390108","volume":"39","author":"X Liu","year":"2005","unstructured":"Liu X, Croft WB (2005) Statistical language modeling for information retrieval. Annu Rev Inf Sci Technol 39(1):1\u201331. https:\/\/doi.org\/10.1002\/ARIS.1440390108","journal-title":"Annu Rev Inf Sci Technol"},{"issue":"9","key":"11405_CR67","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1145\/3560815","volume":"55","author":"P Liu","year":"2023","unstructured":"Liu P, Yuan W, Fu J, Jiang Z, Hayashi H, Neubig G (2023) Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing. ACM Comput Surv 55(9):195\u2013119535. https:\/\/doi.org\/10.1145\/3560815","journal-title":"ACM Comput Surv"},{"key":"11405_CR68","doi-asserted-by":"publisher","unstructured":"Liu Y, Iter D, Xu Y, Wang S, Xu R, Zhu C (2023) G-eval: NLG evaluation using gpt-4 with better human alignment. In: Proceedings of the 2023 Conference on empirical methods in natural language processing, EMNLP 2023, pp. 2511\u20132522. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.153","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"11405_CR69","doi-asserted-by":"publisher","unstructured":"Liu Z, Oguz B, Zhao C, Chang E, Stock P, Mehdad Y, Shi Y, Krishnamoorthi R, Chandra V (2024) Llm-qat: Data-free quantization aware training for large language models. In: Findings of the association for computational linguistics ACL 2024, pp. 467\u2013484. https:\/\/doi.org\/10.18653\/V1\/2024.FINDINGS-ACL.26","DOI":"10.18653\/V1\/2024.FINDINGS-ACL.26"},{"key":"11405_CR70","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) Roberta: a robustly optimized BERT pretraining approach. Preprint at arXiv:1907.11692"},{"key":"11405_CR71","doi-asserted-by":"publisher","unstructured":"Liu A, Swayamdipta S, Smith NA, Choi Y (2022) WANLI: worker and AI collaboration for natural language inference dataset creation. In: Findings of the association for computational linguistics: EMNLP, pp. 6826\u20136847. Association for computational linguistics, Abu Dhabi, United Arab Emirates. https:\/\/doi.org\/10.18653\/v1\/2022.findings-emnlp.508","DOI":"10.18653\/v1\/2022.findings-emnlp.508"},{"key":"11405_CR72","doi-asserted-by":"publisher","unstructured":"Liu Q, Wu X, Wang W, Wang Y, Zhu Y, Zhao X, Tian F, Zheng Y (2025) Llmemb: Large language model can be a good embedding generator for sequential recommendation. In: AAAI-25, Sponsored by the association for the advancement of artificial intelligence, pp. 12183\u201312191. AAAI Press, Philadelphia, PA, USA. https:\/\/doi.org\/10.1609\/AAAI.V39I11.33327","DOI":"10.1609\/AAAI.V39I11.33327"},{"key":"11405_CR73","doi-asserted-by":"publisher","unstructured":"Long Q, Wang W, Pan SJ (2023) Adapt in contexts: Retrieval-augmented domain adaptation via in-context learning. In: Proceedings of the 2023 conference on empirical methods in natural language processing, pp. 6525\u20136542. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.402","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.402"},{"key":"11405_CR74","unstructured":"Lu H, Lam W (2023) EPA: Easy prompt augmentation on large language models via multiple sources and multiple targets. Preprint at arXiv:2309.04725"},{"key":"11405_CR75","doi-asserted-by":"publisher","unstructured":"Lyu X, Min S, Beltagy I, Zettlemoyer L, Hajishirzi H (2023) Z-ICL: zero-shot in-context learning with pseudo-demonstrations. In: Proceedings of the 61st annual meeting of the association for computational linguistics (Volume 1: Long Papers), pp. 2304\u20132317. Association for computational linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/V1\/2023.ACL-LONG.129","DOI":"10.18653\/V1\/2023.ACL-LONG.129"},{"key":"11405_CR76","unstructured":"Maas AL, Daly RE, Pham PT, Huang D, Ng AY, Potts C (2011) Learning word vectors for sentiment analysis. In: Proceedings of the 49th Annual meeting of the association for computational linguistics: human language technologies, pp. 142\u2013150. Association for computational linguistics, Portland, Oregon, USA. https:\/\/aclanthology.org\/P11-1015\/"},{"key":"11405_CR77","doi-asserted-by":"publisher","unstructured":"McAuley JJ, Leskovec J (2013) Hidden factors and hidden topics: understanding rating dimensions with review text. In: Seventh ACM Conference on Recommender Systems, RecSys \u201913, pp. 165\u2013172. ACM, Hong Kong, China. https:\/\/doi.org\/10.1145\/2507157.2507163","DOI":"10.1145\/2507157.2507163"},{"issue":"11","key":"11405_CR78","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1145\/219717.219748","volume":"38","author":"GA Miller","year":"1995","unstructured":"Miller GA (1995) Wordnet: a lexical database for English. Commun ACM 38(11):39\u201341. https:\/\/doi.org\/10.1145\/219717.219748","journal-title":"Commun ACM"},{"key":"11405_CR79","unstructured":"M\u00f6ller T, Reina A, Jayakumar R, Pietsch M (2020) COVID-QA: A question answering dataset for COVID-19. In: Proceedings of the 1st Workshop on NLP for COVID-19 at ACL 2020. Association for computational linguistics, Online. https:\/\/aclanthology.org\/2020.nlpcovid19-acl.18\/"},{"key":"11405_CR80","doi-asserted-by":"publisher","unstructured":"Nguyen HC, Tariq S, Chhetri MB, Vo BQ (2025) Towards effective identification of attack techniques in cyber threat intelligence reports using large language models. In: Companion Proceedings of the ACM on Web Conference 2025, WWW 2025, pp. 942\u2013946. ACM, Sydney, NSW, Australia. https:\/\/doi.org\/10.1145\/3701716.3715469","DOI":"10.1145\/3701716.3715469"},{"key":"11405_CR81","doi-asserted-by":"publisher","unstructured":"Ni J, Qu C, Lu J, Dai Z, \u00c1brego GH, Ma J, Zhao VY, Luan Y, Hall KB, Chang M, Yang Y (2022) Large dual encoders are generalizable retrievers. In: Proceedings of the 2022 conference on empirical methods in natural language processing, EMNLP, pp. 9844\u20139855. Association for Computational Linguistics, United Arab Emirates. https:\/\/doi.org\/10.18653\/V1\/2022.EMNLP-MAIN.669","DOI":"10.18653\/V1\/2022.EMNLP-MAIN.669"},{"key":"11405_CR82","unstructured":"Oh S, Lee SA, Jung W (2023) Data augmentation for neural machine translation using generative language model. Preprint at arXiv:2307.16833"},{"key":"11405_CR83","unstructured":"Ouyang L et al (2022) Training language models to follow instructions with human feedback. In: Advances in neural information processing systems 35: Annual conference on neural information processing systems 2022, New Orleans, LA, USA"},{"key":"11405_CR84","unstructured":"Patel A, Li B, Rasooli MS, Constant N, Raffel C, Callison-Burch C (2023) Bidirectional language models are also few-shot learners. In: The Eleventh International conference on learning representations. OpenReview.net, Kigali, Rwanda"},{"key":"11405_CR85","doi-asserted-by":"publisher","unstructured":"Poth C, Sterz H, Paul I, Purkayastha S, Engl\u00e4nder L, Imhof T, Vulic I, Ruder S, Gurevych I, Pfeiffer J (2023) Adapters: A unified library for parameter-efficient and modular transfer learning. In: Proceedings of the 2023 Conference on empirical methods in natural language processing, EMNLP 2023 - System Demonstrations, pp. 149\u2013160. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-DEMO.13","DOI":"10.18653\/V1\/2023.EMNLP-DEMO.13"},{"key":"11405_CR86","unstructured":"Radford A (2018) Improving language understanding by generative pre-training"},{"key":"11405_CR87","unstructured":"Radford A, Wu J, Child R, Luan D, Amodei D, Sutskever I et al (2019) Language models are unsupervised multitask learners"},{"key":"11405_CR88","first-page":"140","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel C, Shazeer N, Roberts A, Lee K, Narang S, Matena M, Zhou Y, Li W, Liu PJ (2020) Exploring the limits of transfer learning with a unified text-to-text transformer. J Mach Learn Res 21:140\u2013114067","journal-title":"J Mach Learn Res"},{"key":"11405_CR89","doi-asserted-by":"publisher","unstructured":"Rajpurkar P, Zhang J, Lopyrev K, Liang P (2016) Squad: 100, 000+ questions for machine comprehension of text. In: Proceedings of the 2016 conference on empirical methods in natural language processing, EMNLP, pp. 2383\u20132392. The association for computational linguistics, Austin, Texas, USA. https:\/\/doi.org\/10.18653\/v1\/d16-1264","DOI":"10.18653\/v1\/d16-1264"},{"key":"11405_CR90","doi-asserted-by":"publisher","unstructured":"Reimers N, Gurevych I (2019) Sentence-bert: Sentence embeddings using siamese bert-networks. In: Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th International joint conference on natural language processing, EMNLP-IJCNLP, pp. 3980\u20133990. Association for computational linguistics, Hong Kong, China. https:\/\/doi.org\/10.18653\/V1\/D19-1410","DOI":"10.18653\/V1\/D19-1410"},{"key":"11405_CR91","doi-asserted-by":"publisher","unstructured":"Ren Y, Cao Y, Guo P, Fang F, Ma W, Lin Z (2023) Retrieve-and-sample: Document-level event argument extraction via hybrid retrieval augmentation. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL, pp. 293\u2013306. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/V1\/2023.ACL-LONG.17","DOI":"10.18653\/V1\/2023.ACL-LONG.17"},{"key":"11405_CR92","doi-asserted-by":"crossref","unstructured":"Robertson SE, Walker S, Jones S, Hancock-Beaulieu MM, Gatford M et al (1995) Okapi at trec-3. Nist Special Publication Sp 109, 109. http:\/\/trec.nist.gov\/pubs\/trec3\/papers\/city.ps.gz","DOI":"10.6028\/NIST.SP.500-225.city"},{"key":"11405_CR93","doi-asserted-by":"crossref","unstructured":"Rose S, Engel D, Cramer N, Cowley W (2010) Automatic keyword extraction from individual documents. Text mining: applications and theory, 1\u201320","DOI":"10.1002\/9780470689646.ch1"},{"key":"11405_CR94","doi-asserted-by":"publisher","unstructured":"Saad-Falcon J, Khattab O, Santhanam K, Florian R, Franz M, Roukos S, Sil A, Sultan MA, Potts C (2023) UDAPDR: unsupervised domain adaptation via LLM prompting and distillation of rerankers. In: Proceedings of the 2023 conference on empirical methods in natural language processing, EMNLP, pp. 11265\u201311279. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.693","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.693"},{"key":"11405_CR95","doi-asserted-by":"publisher","unstructured":"Saakyan A, Muresan S (2024) ICLEF: in-context learning with expert feedback for explainable style transfer. In: Proceedings of the 62nd Annual meeting of the association for computational linguistics (Volume 1: Long Papers), pp. 16141\u201316163. Association for computational linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.854","DOI":"10.18653\/v1\/2024.acl-long.854"},{"key":"11405_CR96","doi-asserted-by":"crossref","unstructured":"Sahoo P, Singh AK, Saha S, Jain V, Mondal S, Chadha A (2024) A systematic survey of prompt engineering in large language models: techniques and applications. Preprint at arXiv:2402.07927","DOI":"10.1007\/979-8-8688-0569-1_4"},{"key":"11405_CR97","doi-asserted-by":"publisher","unstructured":"Sahu G, Rodr\u00edguez P, Laradji IH, Atighehchian P, V\u00e1zquez D, Bahdanau D (2022) Data augmentation for intent classification with off-the-shelf large language models. In: Proceedings of the 4th Workshop on NLP for Conversational AI, pp. 47\u201357. Association for Computational Linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/V1\/2022.NLP4CONVAI-1.5","DOI":"10.18653\/V1\/2022.NLP4CONVAI-1.5"},{"key":"11405_CR98","doi-asserted-by":"publisher","unstructured":"Sahu G, Vechtomova O, Bahdanau D, Laradji IH (2023) Promptmix: a class boundary augmentation method for large language model distillation. In: Proceedings of the 2023 conference on empirical methods in natural language processing, pp. 5316\u20135327. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.323","DOI":"10.18653\/v1\/2023.emnlp-main.323"},{"issue":"4","key":"11405_CR99","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1108\/eb026562","volume":"29","author":"G Salton","year":"1973","unstructured":"Salton G, Yang C-S (1973) On the specification of term values in automatic indexing. J Doc 29(4):351\u2013372","journal-title":"J Doc"},{"key":"11405_CR100","doi-asserted-by":"publisher","unstructured":"Samuel V, Aynaou H, Chowdhury AG, Ramanan KV, Chadha A (2024) Can llms augment low-resource reading comprehension datasets? opportunities and challenges. In: Proceedings of the 62nd annual meeting of the association for computational linguistics, pp. 411\u2013421. Association for computational linguistics, Bangkok, Thailand. https:\/\/doi.org\/10.18653\/V1\/2024.ACL-SRW.36","DOI":"10.18653\/V1\/2024.ACL-SRW.36"},{"key":"11405_CR101","unstructured":"Sanh V, Debut L, Chaumond J, Wolf T (2019) Distilbert, a distilled version of BERT: smaller, faster, cheaper and lighter. Preprint at arXiv:1910.01108"},{"key":"11405_CR102","doi-asserted-by":"publisher","unstructured":"Schick T, Sch\u00fctze H (2021) It\u2019s not just size that matters: Small language models are also few-shot learners. In: Proceedings of the 2021 Conference of the North American chapter of the association for computational linguistics: Human language technologies, NAACL-HLT, pp. 2339\u20132352. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2021.NAACL-MAIN.185","DOI":"10.18653\/V1\/2021.NAACL-MAIN.185"},{"key":"11405_CR103","unstructured":"Schlegel V, Li H, Wu Y, Subramanian A, Nguyen T, Kashyap AR, Beck D, Zeng X, Batista-Navarro RT, Winkler S, Nenadic G (2023) PULSAR at mediqa-sum 2023: large language models augmented by synthetic dialogue convert patient dialogues to medical records. In: Working Notes of the Conference and Labs of the Evaluation Forum. CEUR Workshop Proceedings, vol. 3497, pp. 1668\u20131679. CEUR-WS.org, Thessaloniki, Greece. https:\/\/ceur-ws.org\/Vol-3497\/paper-138.pdf"},{"key":"11405_CR104","doi-asserted-by":"publisher","unstructured":"Sennrich R, Haddow B, Birch A (2016) Improving neural machine translation models with monolingual data. In: Proceedings of the 54th annual meeting of the association for computational linguistics, ACL 2016. The association for computer linguistics, Berlin, Germany. https:\/\/doi.org\/10.18653\/V1\/P16-1009","DOI":"10.18653\/V1\/P16-1009"},{"key":"11405_CR105","unstructured":"Seo M, Baek J, Thorne J, Hwang SJ (2024) Retrieval-augmented data augmentation for low-resource domain tasks. Preprint at arXiv:2402.13482"},{"key":"11405_CR106","doi-asserted-by":"crossref","unstructured":"Shailya K, Rajpal S, Krishnan GS, Ravindran B (2025) Lext: Towards evaluating trustworthiness of natural language explanations. Preprint at arXiv:2504.06227","DOI":"10.1145\/3715275.3732104"},{"issue":"1","key":"11405_CR107","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1186\/s40537-021-00492-0","volume":"8","author":"C Shorten","year":"2021","unstructured":"Shorten C, Khoshgoftaar TM, Furht B (2021) Text data augmentation for deep learning. J Big Data 8(1):101. https:\/\/doi.org\/10.1186\/s40537-021-00492-0","journal-title":"J Big Data"},{"key":"11405_CR108","doi-asserted-by":"publisher","unstructured":"Shuster K, Komeili M, Adolphs L, Roller S, Szlam A, Weston J (2022) Language models that seek for knowledge: Modular search & generation for dialogue and prompt completion. In: Findings of the association for computational linguistics: EMNLP, pp. 373\u2013393. Association for Computational Linguistics, United Arab Emirates. https:\/\/doi.org\/10.18653\/V1\/2022.FINDINGS-EMNLP.27","DOI":"10.18653\/V1\/2022.FINDINGS-EMNLP.27"},{"key":"11405_CR109","doi-asserted-by":"publisher","unstructured":"Socher R, Perelygin A, Wu J, Chuang J, Manning CD, Ng AY, Potts C (2013) Recursive deep models for semantic compositionality over a sentiment treebank. In: Proceedings of the 2013 Conference on empirical methods in natural language processing, EMNLP, pp. 1631\u20131642. ACL, Seattle, Washington, USA. https:\/\/doi.org\/10.18653\/v1\/d13-1170","DOI":"10.18653\/v1\/d13-1170"},{"key":"11405_CR110","unstructured":"Song K, Tan X, Qin T, Lu J, Liu T (2020) Mpnet: Masked and permuted pre-training for language understanding. In: Advances in Neural Information Processing Systems 33: Annual conference on neural information processing Systems 2020, NeurIPS 2020, Virtual Event. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/c3a690be93aa602ee2dc0ccab5b7b67e-Abstract.html"},{"key":"11405_CR111","doi-asserted-by":"publisher","unstructured":"Sottana A, Liang B, Zou K, Yuan Z (2023) Evaluation metrics in the era of GPT-4: reliably evaluating large language models on sequence to sequence tasks. In: Proceedings of the 2023 Conference on empirical methods in natural language processing, EMNLP, pp. 8776\u20138788. Association for computational linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.543","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.543"},{"key":"11405_CR112","doi-asserted-by":"publisher","DOI":"10.1016\/J.KNOSYS.2024.111740","volume":"294","author":"L Souza Silva","year":"2024","unstructured":"Souza Silva L, Barbosa L (2024) Improving dense retrieval models with LLM augmented data for dataset search. Knowl Based Syst 294:111740. https:\/\/doi.org\/10.1016\/J.KNOSYS.2024.111740","journal-title":"Knowl Based Syst"},{"key":"11405_CR113","doi-asserted-by":"publisher","unstructured":"Szymanski A, Ziems N, Eicher-Miller HA, Li TJ-J, Jiang M, Metoyer RA (2025) Limitations of the llm-as-a-judge approach for evaluating llm outputs in expert knowledge tasks. In: Proceedings of the 30th international conference on intelligent user interfaces, pp. 952\u2013966. https:\/\/doi.org\/10.1145\/3708359.3712091","DOI":"10.1145\/3708359.3712091"},{"key":"11405_CR114","doi-asserted-by":"publisher","unstructured":"Thakur N, Reimers N, Daxenberger J, Gurevych I (2021) Augmented SBERT: data augmentation method for improving bi-encoders for pairwise sentence scoring tasks. In: Proceedings of the 2021 Conference of the North American chapter of the association for computational linguistics: human language technologies, pp. 296\u2013310. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2021.NAACL-MAIN.28","DOI":"10.18653\/V1\/2021.NAACL-MAIN.28"},{"key":"11405_CR115","unstructured":"Thulke D, Daheim N, Dugast C, Ney H (2021) Efficient retrieval augmented generation from unstructured knowledge for task-oriented dialog. Preprint at arXiv:2102.04643"},{"key":"11405_CR116","unstructured":"Tian J, Chen H, Xu G, Yan M, Gao X, Zhang J, Li C, Liu J, Xu W, Xu H, Qian Q, Wang W, Ye Q, Zhang J, Zhang J, Huang F, Zhou J (2023) ChatPLUG: open-domain generative dialogue system with internet-augmented instruction tuning for digital human. Preprint at arXiv:2304.07849"},{"key":"11405_CR117","doi-asserted-by":"crossref","unstructured":"Tjong Kim Sang EF, De Meulder F (2003) Introduction to the CoNLL-2003 shared task: Language-independent named entity recognition. In: Proceedings of the Seventh conference on natural language learning at HLT-NAACL 2003, pp. 142\u2013147. https:\/\/aclanthology.org\/W03-0419\/","DOI":"10.3115\/1119176.1119195"},{"key":"11405_CR118","unstructured":"Touvron H et al (2023) Llama 2: open foundation and fine-tuned chat models. Preprint at arXiv:2307.09288"},{"key":"11405_CR119","unstructured":"Touvron H, Lavril T, Izacard G, Martinet X, Lachaux M, Lacroix T, Rozi\u00e8re B, Goyal N, Hambro E, Azhar F, Rodriguez A, Joulin A, Grave E, Lample G (2023) LLaMA: open and efficient foundation language models. Preprint at arXiv:2302.13971"},{"key":"11405_CR120","unstructured":"Ubani S, Polat SO, Nielsen R (2023) ZeroShotDataAug: generating and augmenting training data with ChatGPT. Preprint at arXiv:2304.14334"},{"key":"11405_CR121","doi-asserted-by":"publisher","unstructured":"Van H, Yadav V, Surdeanu M (2021) Cheap and good? simple and effective data augmentation for low resource machine reading. In: SIGIR \u201921: The 44th International ACM SIGIR conference on research and development in information retrieval, virtual event, pp. 2116\u20132120. ACM, Canada. https:\/\/doi.org\/10.1145\/3404835.3463099","DOI":"10.1145\/3404835.3463099"},{"key":"11405_CR122","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Advances in neural information processing systems 30: annual conference on neural information processing systems, USA, pp. 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"11405_CR123","doi-asserted-by":"crossref","unstructured":"Voorhees EM, Tice DM (1999) The TREC-8 question answering track evaluation. In: Proceedings of The Eighth Text retrieval conference, TREC. NIST Special Publication, vol. 500-246. National Institute of Standards and Technology (NIST), Gaithersburg, Maryland, USA. http:\/\/trec.nist.gov\/pubs\/trec8\/papers\/qa8.pdf","DOI":"10.6028\/NIST.SP.500-246.qa-overview"},{"key":"11405_CR124","doi-asserted-by":"publisher","DOI":"10.1016\/J.ARTINT.2023.103874","volume":"319","author":"A Wang","year":"2023","unstructured":"Wang A, Song L, Liu Q, Mi H, Wang L, Tu Z, Su J, Yu D (2023) Search-engine-augmented dialogue response generation with cheaply supervised query production. Artif Intell 319:103874. https:\/\/doi.org\/10.1016\/J.ARTINT.2023.103874","journal-title":"Artif Intell"},{"key":"11405_CR125","doi-asserted-by":"publisher","unstructured":"Wang M, Adel H, Lange L, Str\u00f6tgen J, Sch\u00fctze H (2024) Rehearsal-free modular and compositional continual learning for language models. In: Proceedings of the 2024 Conference of the North American chapter of the association for computational linguistics: human language technologies: Short Papers, NAACL, pp. 469\u2013480. Association for Computational Linguistics, Mexico City, Mexico. https:\/\/doi.org\/10.18653\/V1\/2024.NAACL-SHORT.39","DOI":"10.18653\/V1\/2024.NAACL-SHORT.39"},{"key":"11405_CR126","unstructured":"Wang H, Huang W, Deng Y, Wang R, Wang Z, Wang Y, Mi F, Pan JZ, Wong K (2024) UniMS-RAG: a unified multi-source retrieval-augmented generation for personalized dialogue systems. Preprint at arXiv:2401.13256"},{"key":"11405_CR127","unstructured":"Wang B, Komatsuzaki A (2021) GPT-J-6B: A 6 billion parameter autoregressive language model"},{"key":"11405_CR128","doi-asserted-by":"publisher","unstructured":"Wei JW, Zou K (2019) EDA: easy data augmentation techniques for boosting performance on text classification tasks. In: Inui K, Jiang J, Ng V, Wan X (eds.) Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing, EMNLP-IJCNLP, pp. 6381\u20136387. Association for Computational Linguistics, Hong Kong, China. https:\/\/doi.org\/10.18653\/v1\/D19-1670","DOI":"10.18653\/v1\/D19-1670"},{"key":"11405_CR129","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1613\/JAIR.1.17809","volume":"82","author":"C Wei","year":"2025","unstructured":"Wei C, Duan K, Zhuo S, Wang H, Huang S, Liu J (2025) Enhanced recommendation systems with retrieval-augmented large language model. J Artif Intell Res 82:1147\u20131173. https:\/\/doi.org\/10.1613\/JAIR.1.17809","journal-title":"J Artif Intell Res"},{"key":"11405_CR130","doi-asserted-by":"publisher","unstructured":"Whitehouse C, Choudhury M, Aji AF (2023) Llm-powered data augmentation for enhanced cross-lingual performance. In: Proceedings of the 2023 conference on empirical methods in natural language processing, pp. 671\u2013686. Association for Computational Linguistics, Singapore. https:\/\/doi.org\/10.18653\/V1\/2023.EMNLP-MAIN.44","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.44"},{"key":"11405_CR131","doi-asserted-by":"publisher","unstructured":"Wu Z, Galley M, Brockett C, Zhang Y, Gao X, Quirk C, Koncel-Kedziorski R, Gao J, Hajishirzi H, Ostendorf M, Dolan B (2021) A controllable model of grounded response generation. In: Thirty-Fifth AAAI conference on artificial intelligence, AAAI 2021, thirty-third conference on innovative applications of artificial intelligence, IAAI 2021, The Eleventh Symposium on Educational Advances in Artificial Intelligence, EAAI, pp. 14085\u201314093. AAAI Press, Virtual Event. https:\/\/doi.org\/10.1609\/AAAI.V35I16.17658","DOI":"10.1609\/AAAI.V35I16.17658"},{"key":"11405_CR132","doi-asserted-by":"publisher","unstructured":"Wu H, Lei C, Sun X, Wang P-S, Chen Q, Cheng K-T, Lin S, Wu Z (2023) Randomized quantization: a generic augmentation for data agnostic self-supervised learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16305\u201316316. https:\/\/doi.org\/10.1109\/ICCV51070.2023.01494","DOI":"10.1109\/ICCV51070.2023.01494"},{"key":"11405_CR133","unstructured":"Xiong L, Xiong C, Li Y, Tang K, Liu J, Bennett PN, Ahmed J, Overwijk A (2021) Approximate nearest neighbor negative contrastive learning for dense text retrieval. In: 9th International Conference on Learning Representations, ICLR. OpenReview.net, Austria. https:\/\/openreview.net\/forum?id=zeFrfgyZln"},{"issue":"1","key":"11405_CR134","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1109\/MIS.2024.3508432","volume":"40","author":"L Xu","year":"2025","unstructured":"Xu L, Xie H, Qin SJ, Wang FL, Tao X (2025) Exploring chatgpt-based augmentation strategies for contrastive aspect-based sentiment analysis. IEEE Intell Syst 40(1):69\u201376. https:\/\/doi.org\/10.1109\/MIS.2024.3508432","journal-title":"IEEE Intell Syst"},{"key":"11405_CR135","unstructured":"Xu X, Li M, Tao C, Shen T, Cheng R, Li J, Xu C, Tao D, Zhou T (2024) A survey on knowledge distillation of large language models. Preprint at arXiv:2402.13116"},{"key":"11405_CR136","doi-asserted-by":"publisher","unstructured":"Xu D, Li X, Zhang Z, Lin Z, Zhu Z, Zheng Z, Wu X, Zhao X, Xu T, Chen E (2025) Harnessing large language models for knowledge graph question answering via adaptive multi-aspect retrieval-augmentation. In: AAAI-25, Sponsored by the association for the advancement of artificial intelligence, pp. 25570\u201325578. AAAI Press, Philadelphia, PA, USA. https:\/\/doi.org\/10.1609\/AAAI.V39I24.34747","DOI":"10.1609\/AAAI.V39I24.34747"},{"key":"11405_CR137","doi-asserted-by":"publisher","unstructured":"Xu H, Niu Y, Wen Y, Yuan X (2025a) Compress and mix: Advancing efficient taxonomy completion with large language models. In: Proceedings of the ACM on Web Conference 2025, WWW 2025, pp. 4239\u20134249. ACM, Sydney, NSW, Australia. https:\/\/doi.org\/10.1145\/3696410.3714690","DOI":"10.1145\/3696410.3714690"},{"key":"11405_CR138","doi-asserted-by":"crossref","unstructured":"Xu H, Zhao N, Yang L, Zhao S, Deng S, Wang M, Hooi B, Oo N, Chen H, Zhang N (2025b) Relearn: Unlearning via learning for large language models. In: Proceedings of the 63rd annual meeting of the association for computational linguistics (Volume 1: Long Papers), ACL 2025, pp. 5967\u20135987. Association for Computational Linguistics, Vienna, Austria. https:\/\/aclanthology.org\/2025.acl-long.297\/","DOI":"10.18653\/v1\/2025.acl-long.297"},{"key":"11405_CR139","doi-asserted-by":"publisher","unstructured":"Yang Y, Malaviya C, Fernandez J, Swayamdipta S, Bras RL, Wang J, Bhagavatula C, Choi Y, Downey D (2020) G-daug: Generative data augmentation for commonsense reasoning. In: Findings of the Association for Computational Linguistics: EMNLP. Findings of ACL, vol. EMNLP 2020, pp. 1008\u20131025. Association for Computational Linguistics, Online. https:\/\/doi.org\/10.18653\/V1\/2020.FINDINGS-EMNLP.90","DOI":"10.18653\/V1\/2020.FINDINGS-EMNLP.90"},{"key":"11405_CR140","doi-asserted-by":"publisher","unstructured":"Yang Z, Qi P, Zhang S, Bengio Y, Cohen WW, Salakhutdinov R, Manning CD (2018) Hotpotqa: A dataset for diverse, explainable multi-hop question answering. In: Proceedings of the 2018 conference on empirical methods in natural language processing, pp. 2369\u20132380. Association for Computational Linguistics, Brussels, Belgium. https:\/\/doi.org\/10.18653\/v1\/d18-1259","DOI":"10.18653\/v1\/d18-1259"},{"key":"11405_CR141","doi-asserted-by":"publisher","unstructured":"Yang D, Rao J, Chen K, Guo X, Zhang Y, Yang J, Zhang Y (2024) IM-RAG: multi-round retrieval-augmented generation through learning inner monologues. In: Proceedings of the 47th International ACM SIGIR conference on research and development in information retrieval, SIGIR, pp. 730\u2013740. ACM, Washington DC, USA. https:\/\/doi.org\/10.1145\/3626772.3657760","DOI":"10.1145\/3626772.3657760"},{"key":"11405_CR142","unstructured":"Yang R, Wu T, Wang J, Hu P, Wu Y-C, Wong N, Yang Y (2024) Llm-neo: Parameter efficient knowledge distillation for large language models. Preprint at arXiv:2411.06839"},{"key":"11405_CR143","unstructured":"Yao S, Zhao J, Yu D, Du N, Shafran I, Narasimhan KR, Cao Y (2023) React: synergizing reasoning and acting in language models. In: The Eleventh International Conference on Learning Representations, ICLR. OpenReview.net, Kigali, Rwanda. https:\/\/openreview.net\/forum?id=WE_vluYUL-X"},{"key":"11405_CR144","doi-asserted-by":"publisher","unstructured":"Ye S, Hwang H, Yang S, Yun H, Kim Y, Seo M (2024) Investigating the effectiveness of task-agnostic prefix prompt for instruction following. In: Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, Thirty-Sixth conference on innovative applications of artificial intelligence, IAAI 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence, EAAI, pp. 19386\u201319394. AAAI Press, Vancouver, Canada. https:\/\/doi.org\/10.1609\/AAAI.V38I17.29909","DOI":"10.1609\/AAAI.V38I17.29909"},{"key":"11405_CR145","unstructured":"Ye J, Xu N, Wang Y, Zhou J, Zhang Q, Gui T, Huang X (2024) LLM-DA: data augmentation via large language models for few-shot named entity recognition. Preprint at arXiv:2402.14568"},{"key":"11405_CR146","doi-asserted-by":"publisher","unstructured":"Yoo KM, Park D, Kang J, Lee S, Park W (2021) Gpt3mix: Leveraging large-scale language models for text augmentation. In: Findings of the association for computational linguistics: EMNLP, pp. 2225\u20132239. Association for computational linguistics, Dominican Republic. https:\/\/doi.org\/10.18653\/V1\/2021.FINDINGS-EMNLP.192","DOI":"10.18653\/V1\/2021.FINDINGS-EMNLP.192"},{"key":"11405_CR147","unstructured":"Yuan J, Tang R, Jiang X, Hu X (2023) Large Language Models for Healthcare Data augmentation: an example on patient-trial matching. Preprint at arXiv:2303.16756"},{"issue":"3","key":"11405_CR148","doi-asserted-by":"publisher","first-page":"867","DOI":"10.1162\/COLI_A_00515","volume":"50","author":"M Zhang","year":"2024","unstructured":"Zhang M, Jiang G, Liu S, Chen J, Zhang M (2024) Llm-assisted data augmentation for Chinese dialogue-level dependency parsing. Comput Linguistics 50(3):867\u2013891. https:\/\/doi.org\/10.1162\/COLI_A_00515","journal-title":"Comput Linguistics"},{"key":"11405_CR149","unstructured":"Zhang Z, Hou Y, Gong C, Li Z (2025) Data augmentation for cross-domain parsing via lightweight LLM generation and tree hybridization. In: Proceedings of the 31st International conference on computational linguistics, COLING 2025, pp. 11235\u201311247. Association for Computational Linguistics, Abu Dhabi, UAE. https:\/\/aclanthology.org\/2025.coling-main.744\/"},{"key":"11405_CR150","doi-asserted-by":"publisher","unstructured":"Zhang Y, Sun S, Gao X, Fang Y, Brockett C, Galley M, Gao J, Dolan B (2022) Retgen: A joint framework for retrieval and grounded text generation modeling. In: Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI 2022, thirty-fourth conference on innovative applications of artificial intelligence, IAAI 2022, The Twelveth Symposium on Educational Advances in Artificial Intelligence, EAAI, pp. 11739\u201311747. AAAI Press, Virtual Event. https:\/\/doi.org\/10.1609\/AAAI.V36I10.21429","DOI":"10.1609\/AAAI.V36I10.21429"},{"key":"11405_CR151","doi-asserted-by":"publisher","unstructured":"Zhang C, Zhang L, Wu J, He Y, Zhou D (2025) Causal prompting: Debiasing large language model prompting based on front-door adjustment. In: AAAI-25, Sponsored by the Association for the Advancement of Artificial Intelligence, pp. 25842\u201325850. AAAI Press, Philadelphia, PA, USA. https:\/\/doi.org\/10.1609\/AAAI.V39I24.34777","DOI":"10.1609\/AAAI.V39I24.34777"},{"key":"11405_CR152","unstructured":"Zhang X, Zhao JJ, LeCun Y (2015) Character-level convolutional networks for text classification. In: Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7-12, 2015, Montreal, Quebec, Canada, pp. 649\u2013657. https:\/\/proceedings.neurips.cc\/paper\/2015\/hash\/250cf8b51c773f3f8dc8b4be867a9a02-Abstract.html"},{"key":"11405_CR153","unstructured":"Zhao K, Zhao M (2024) Self-supervised quantization-aware knowledge distillation. In: International conference on artificial intelligence and statistics. proceedings of machine learning research, vol. 238, pp. 4375\u20134383. PMLR, Palau de Congressos, Valencia, Spain. https:\/\/proceedings.mlr.press\/v238\/zhao24d.html"},{"key":"11405_CR154","unstructured":"Zhen L et al (2023) Judging llm-as-a-judge with mt-bench and chatbot arena. In: Advances in Neural Information Processing Systems 36: Annual conference on neural information processing systems 2023, NeurIPS 2023, New Orleans, LA, USA. http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/91f18a1287b398d378ef22505bf41832-Abstract-Datasets_and_Benchmarks.html"},{"key":"11405_CR155","doi-asserted-by":"publisher","unstructured":"Zheng C, Sabour S, Wen J, Zhang Z, Huang M (2023) Augesc: Dialogue augmentation with large language models for emotional support conversation. In: Findings of the association for computational linguistics: ACL, pp. 1552\u20131568. Association for Computational Linguistics, Toronto, Canada. https:\/\/doi.org\/10.18653\/V1\/2023.FINDINGS-ACL.99","DOI":"10.18653\/V1\/2023.FINDINGS-ACL.99"},{"key":"11405_CR156","unstructured":"Zheng Z, Song X, Liu C (2024) Mixllm: LLM quantization with global mixed-precision between output-features and highly-efficient system design. Preprint at arXiv:2412.14590"},{"key":"11405_CR157","unstructured":"Zhou Y, Guo C, Wang X, Chang Y, Wu Y (2024) A survey on data augmentation in large model era. Preprint at arXiv:2401.15422"},{"key":"11405_CR158","doi-asserted-by":"publisher","unstructured":"Zhou H, Huang H, Long Y, Xu B, Zhu C, Cao H, Yang M, Zhao T (2024) Mitigating the bias of large language model evaluation. In: Chinese Computational Linguistics - 23rd China National Conference, CCL. Lecture Notes in Computer Science, vol. 14761, pp. 451\u2013462. Springer, Taiyuan, China. https:\/\/doi.org\/10.1007\/978-981-97-8367-0_27","DOI":"10.1007\/978-981-97-8367-0_27"},{"key":"11405_CR159","doi-asserted-by":"publisher","unstructured":"Zhou J, Zheng Y, Tang J, Jian L, Yang Z (2022) Flipda: Effective and robust data augmentation for few-shot learning. In: Proceedings of the 60th annual meeting of the association for computational linguistics (Volume 1: Long Papers), ACL, pp. 8646\u20138665. Association for Computational Linguistics, Dublin, Ireland. https:\/\/doi.org\/10.18653\/V1\/2022.ACL-LONG.592","DOI":"10.18653\/V1\/2022.ACL-LONG.592"},{"key":"11405_CR160","doi-asserted-by":"publisher","first-page":"1556","DOI":"10.1162\/tacl_a_00704","volume":"12","author":"X Zhu","year":"2024","unstructured":"Zhu X, Li J, Liu Y, Ma C, Wang W (2024) A survey on model compression for large language models. Trans Assoc Comput Linguistics 12:1556\u20131577. https:\/\/doi.org\/10.1162\/tacl_a_00704","journal-title":"Trans Assoc Comput Linguistics"},{"key":"11405_CR161","doi-asserted-by":"publisher","unstructured":"Zhu Y, Nie J, Dou Z, Ma Z, Zhang X, Du P, Zuo X, Jiang H (2021) Contrastive learning of user behavior sequence for context-aware document ranking. In: CIKM \u201921: The 30th ACM international conference on information and knowledge management, pp. 2780\u20132791. ACM, Queensland, Australia. https:\/\/doi.org\/10.1145\/3459637.3482243","DOI":"10.1145\/3459637.3482243"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11405-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-025-11405-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11405-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T03:09:30Z","timestamp":1769483370000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-025-11405-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,11]]},"references-count":161,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,1]]}},"alternative-id":["11405"],"URL":"https:\/\/doi.org\/10.1007\/s10462-025-11405-5","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,11]]},"assertion":[{"value":"24 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"35"}}