{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T23:44:33Z","timestamp":1786146273304,"version":"3.56.0"},"reference-count":377,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:00:00Z","timestamp":1767312000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,1,6]],"date-time":"2026-01-06T00:00:00Z","timestamp":1767657600000},"content-version":"vor","delay-in-days":4,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23B2048"],"award-info":[{"award-number":["U23B2048"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402011"],"award-info":[{"award-number":["62402011"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Data Sci. Eng."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s41019-025-00335-5","type":"journal-article","created":{"date-parts":[[2026,1,3]],"date-time":"2026-01-03T05:16:32Z","timestamp":1767417392000},"page":"1-29","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":97,"title":["Retrieval-Augmented Generation for AI-Generated Content: A Survey"],"prefix":"10.1007","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1436-6660","authenticated-orcid":false,"given":"Penghao","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hailin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinhan","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhengren","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunteng","family":"Geng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fangcheng","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ling","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wentao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bin","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"335_CR1","unstructured":"Brown TB, Mann B et\u00a0al (2020) Language models are few-shot learners. In: NeurIPS"},{"key":"335_CR2","unstructured":"Chen M, Tworek J et\u00a0al (2021) Evaluating large language models trained on code. arXiv:2107.03374"},{"key":"335_CR3","unstructured":"OpenAI: GPT-4 technical report. arXiv:2303.08774 (2023)"},{"key":"335_CR4","unstructured":"Touvron H et\u00a0al (2023) Llama: open and efficient foundation language models. arXiv:2302.13971"},{"key":"335_CR5","unstructured":"Touvron H et\u00a0al (2023) Llama 2: open foundation and fine-tuned chat models. arXiv:2307.09288"},{"key":"335_CR6","unstructured":"Rozi\u00e8re B, Gehring J et\u00a0al (2023) Code llama: open foundation models for code. arXiv:2308.12950"},{"key":"335_CR7","unstructured":"Ramesh A, Pavlov M, Goh G et\u00a0al (2021) Zero-shot text-to-image generation. In: ICML"},{"key":"335_CR8","unstructured":"Ramesh A, Dhariwal P, Nichol A et\u00a0al (2022) Hierarchical text-conditional image generation with CLIP latents. arXiv:2204.06125"},{"issue":"3","key":"335_CR9","first-page":"8","volume":"2","author":"J Betker","year":"2023","unstructured":"Betker J, Goh G, Jing L et al (2023) Improving image generation with better captions. Comput Sci 2(3):8","journal-title":"Comput Sci"},{"key":"335_CR10","doi-asserted-by":"crossref","unstructured":"Rombach R, Blattmann A, Lorenz D et\u00a0al (2022) High-resolution image synthesis with latent diffusion models. In: IEEE\/CVF","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"335_CR11","unstructured":"OpenAI (2024) Video generation models as world simulators. https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"issue":"8","key":"335_CR12","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"335_CR13","unstructured":"Vaswani A, Shazeer N, Parmar N et\u00a0al (2017) Attention is all you need. In: NeurIPS"},{"issue":"11","key":"335_CR14","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow I, Pouget-Abadie J et al (2020) Generative adversarial networks. CACM 63(11):139\u2013144","journal-title":"CACM"},{"key":"335_CR15","unstructured":"Devlin J, Chang M et\u00a0al (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: NAACL-HLT"},{"key":"335_CR16","first-page":"140","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel C, Shazeer N, Roberts A et al (2020) Exploring the limits of transfer learning with a unified text-to-text transformer. JMLR 21:140","journal-title":"JMLR"},{"issue":"120","key":"335_CR17","first-page":"1","volume":"23","author":"W Fedus","year":"2022","unstructured":"Fedus W, Zoph B, Shazeer N (2022) Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. JMLR 23(120):1\u201339","journal-title":"JMLR"},{"key":"335_CR18","unstructured":"Kaplan J, McCandlish S et\u00a0al (2020) Scaling laws for neural language models. https:\/\/arxiv.org\/abs\/2001.08361"},{"issue":"4","key":"335_CR19","first-page":"333","volume":"3","author":"SE Robertson","year":"2009","unstructured":"Robertson SE, Zaragoza H (2009) The probabilistic relevance framework: BM25 and beyond. FTIR 3(4):333\u2013389","journal-title":"FTIR"},{"key":"335_CR20","doi-asserted-by":"crossref","unstructured":"Karpukhin V, Oguz B, Min S et\u00a0al (2020) Dense passage retrieval for open-domain question answering. In: EMNLP","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"issue":"3","key":"335_CR21","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1109\/TBDATA.2019.2921572","volume":"7","author":"J Johnson","year":"2021","unstructured":"Johnson J, Douze M, J\u00e9gou H (2021) Billion-scale similarity search with GPUs. IEEE Trans Big Data 7(3):535\u2013547","journal-title":"IEEE Trans Big Data"},{"key":"335_CR22","unstructured":"Chen Q, Zhao B, Wang H et\u00a0al (2021) SPANN: highly-efficient billion-scale approximate nearest neighborhood search. In: NeurIPS"},{"issue":"2","key":"335_CR23","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1145\/1348246.1348248","volume":"40","author":"R Datta","year":"2008","unstructured":"Datta R, Joshi D, Li J et al (2008) Image retrieval: ideas, influences, and trends of the new age. CSUR 40(2):5","journal-title":"CSUR"},{"key":"335_CR24","unstructured":"Radford A, Kim JW, Hallacy C et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: ICML"},{"key":"335_CR25","doi-asserted-by":"crossref","unstructured":"Feng Z, Guo D et\u00a0al (2020) Codebert: a pre-trained model for programming and natural languages. In: EMNLP findings","DOI":"10.18653\/v1\/2020.findings-emnlp.139"},{"key":"335_CR26","doi-asserted-by":"crossref","unstructured":"Wu Y, Chen K, Zhang T et\u00a0al (2023) Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation. In: ICASSP","DOI":"10.1109\/ICASSP49357.2023.10095969"},{"key":"335_CR27","doi-asserted-by":"crossref","unstructured":"Mallen A, Asai A, Zhong V et\u00a0al (2023) When not to trust language models: investigating effectiveness of parametric and non-parametric memories. In: ACL","DOI":"10.18653\/v1\/2023.acl-long.546"},{"key":"335_CR28","unstructured":"Carlini N, Tram\u00e8r F et\u00a0al (2021) Extracting training data from large language models. In: USENIX"},{"key":"335_CR29","unstructured":"Kang M, G\u00fcrel NM et\u00a0al (2024) C-RAG: certified generation risks for retrieval-augmented language models. arXiv:2402.03181"},{"key":"335_CR30","unstructured":"Izacard G, Lewis P et\u00a0al (2022) Atlas: few-shot learning with retrieval augmented language models. arXiv:2208.03299"},{"key":"335_CR31","unstructured":"Wu Y, Rabe MN et\u00a0al (2022) Memorizing transformers. In: ICLR"},{"key":"335_CR32","unstructured":"He Z, Zhong Z et\u00a0al (2023) REST: retrieval-based speculative decoding. arxiv:2311.08252"},{"key":"335_CR33","unstructured":"Guu K, Lee K et\u00a0al (2020) REALM: retrieval-augmented language model pre-training. In: ICML"},{"key":"335_CR34","unstructured":"Lewis PSH, Perez E et\u00a0al (2020) Retrieval-augmented generation for knowledge-intensive NLP tasks. In: NeurIPS"},{"key":"335_CR35","doi-asserted-by":"crossref","unstructured":"Izacard G, Grave E (2021) Leveraging passage retrieval with generative models for open domain question answering. In: EACL","DOI":"10.18653\/v1\/2021.eacl-main.74"},{"key":"335_CR36","unstructured":"Borgeaud S, Mensch A et\u00a0al (2022) Improving language models by retrieving from trillions of tokens. In: ICML"},{"key":"335_CR37","unstructured":"Khandelwal U, Levy O, Jurafsky D et\u00a0al (2020) Generalization through memorization: nearest neighbor language models. In: ICLR"},{"key":"335_CR38","doi-asserted-by":"crossref","unstructured":"He J, Neubig G, Berg-Kirkpatrick T (2021) Efficient nearest neighbor language models. In: EMNLP","DOI":"10.18653\/v1\/2021.emnlp-main.461"},{"key":"335_CR39","unstructured":"zilliztech: GPTCache. https:\/\/github.com\/zilliztech\/GPTCache"},{"key":"335_CR40","doi-asserted-by":"crossref","unstructured":"Parvez MR, Ahmad WU et\u00a0al (2021) Retrieval augmented code generation and summarization. In: EMNLP findings","DOI":"10.18653\/v1\/2021.findings-emnlp.232"},{"key":"335_CR41","doi-asserted-by":"crossref","unstructured":"Ahmad WU, Chakraborty S, Ray B et\u00a0al (2021) Unified pre-training for program understanding and generation. In: NAACL-HLT","DOI":"10.18653\/v1\/2021.naacl-main.211"},{"key":"335_CR42","unstructured":"Zhou S, Alon U, Xu FF et\u00a0al (2023) Docprompting: generating code by retrieving the docs. In: ICLR"},{"key":"335_CR43","unstructured":"Koizumi Y, Ohishi Y et\u00a0al (2020) Audio captioning using pre-trained large-scale language model guided by audio-based similar caption retrieval. arXiv:2012.07331"},{"key":"335_CR44","unstructured":"Huang R, Huang J, Yang D et\u00a0al (2023) Make-an-audio: text-to-audio generation with prompt-enhanced diffusion models. In: ICML"},{"key":"335_CR45","doi-asserted-by":"crossref","unstructured":"Tseng H-Y, Lee H-Y et\u00a0al (2020) Retrievegan: image synthesis via differentiable patch retrieval. In: ECCV","DOI":"10.1007\/978-3-030-58598-3_15"},{"key":"335_CR46","doi-asserted-by":"crossref","unstructured":"Sarto S, Cornia M, Baraldi L, Cucchiara R (2022) Retrieval-augmented transformer for image captioning. In: CBMI","DOI":"10.1145\/3549555.3549585"},{"key":"335_CR47","doi-asserted-by":"crossref","unstructured":"Ramos R et\u00a0al (2023) Smallcap: lightweight image captioning prompted with retrieval augmentation. In: CVPR","DOI":"10.1109\/CVPR52729.2023.00278"},{"issue":"1s","key":"335_CR48","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1145\/3539225","volume":"19","author":"J Chen","year":"2023","unstructured":"Chen J, Pan Y, Li Y et al (2023) Retrieval augmented convolutional encoder-decoder networks for video captioning. TOMCCAP 19(1s):48\u201314824","journal-title":"TOMCCAP"},{"key":"335_CR49","doi-asserted-by":"crossref","unstructured":"Xu J, Huang Y et\u00a0al. (2024) Retrieval-augmented egocentric video captioning. arXiv:2401.00789","DOI":"10.1109\/CVPR52733.2024.01284"},{"key":"335_CR50","unstructured":"Seo J, Hong S et\u00a0al (2024) Retrieval-augmented score distillation for text-to-3d generation. arXiv:2402.02972"},{"key":"335_CR51","doi-asserted-by":"crossref","unstructured":"Zhang M, Guo X et\u00a0al (2023) Remodiffuse: retrieval-augmented motion diffusion model. In: ICCV","DOI":"10.1109\/ICCV51070.2023.00040"},{"key":"335_CR52","unstructured":"Hu X, Wu X, Shu Y, Qu Y (2022) Logical form generation via multi-task learning for complex question answering over knowledge bases. In: COLING"},{"key":"335_CR53","doi-asserted-by":"crossref","unstructured":"Huang X, Kim J, Zou B (2021) Unseen entity handling in complex question answering over knowledge base via language generation. In: EMNLP findings","DOI":"10.18653\/v1\/2021.findings-emnlp.50"},{"key":"335_CR54","doi-asserted-by":"crossref","unstructured":"Das R, Zaheer M, Thai D et\u00a0al (2021) Case-based reasoning for natural language queries over knowledge bases. In: EMNLP","DOI":"10.18653\/v1\/2021.emnlp-main.755"},{"key":"335_CR55","unstructured":"Wang Z, Nie W, Qiao Z et\u00a0al (2022) Retrieval-based controllable molecule generation. In: ICLR"},{"issue":"2","key":"335_CR56","doi-asserted-by":"publisher","first-page":"075","DOI":"10.1093\/bioinformatics\/btae075","volume":"40","author":"Q Jin","year":"2024","unstructured":"Jin Q, Yang Y, Chen Q, Lu Z (2024) Genegpt: augmenting large language models with domain tools for improved access to biomedical information. Bioinformatics 40(2):075","journal-title":"Bioinformatics"},{"key":"335_CR57","unstructured":"Li H, Su Y et\u00a0al (2022) A survey on retrieval-augmented text generation. arxiv:2202.01110"},{"key":"335_CR58","unstructured":"Asai A, Min S, Zhong Z, Chen D (2023) ACL 2023 tutorial: retrieval-based language models and applications. In: ACL 2023"},{"key":"335_CR59","unstructured":"Gao Y, Xiong Y et\u00a0al (2023) Retrieval-augmented generation for large language models: a survey. arxiv:2312.10997"},{"key":"335_CR60","doi-asserted-by":"crossref","unstructured":"Zhao R et\u00a0al (2023) Retrieving multimodal information for augmented generation: a survey. In: EMNLP","DOI":"10.18653\/v1\/2023.findings-emnlp.314"},{"key":"335_CR61","doi-asserted-by":"crossref","unstructured":"Fan W, Ding Y, Ning L, Wang S, Li H, Yin D, Chua T-S, Li Q (2024) A survey on rag meeting LLMs: towards retrieval-augmented large language models. In: Proceedings of the 30th ACM SIGKDD conference on knowledge discovery and data mining, pp 6491\u20136501","DOI":"10.1145\/3637528.3671470"},{"key":"335_CR62","doi-asserted-by":"crossref","unstructured":"Chen J, Guo H, Yi K et\u00a0al (2022) Visualgpt: data-efficient adaptation of pretrained language models for image captioning. In: CVPR","DOI":"10.1109\/CVPR52688.2022.01750"},{"issue":"6","key":"335_CR63","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1145\/3530811","volume":"55","author":"Y Tay","year":"2023","unstructured":"Tay Y, Dehghani M et al (2023) Efficient transformers: a survey. CSUR 55(6):109","journal-title":"CSUR"},{"issue":"8","key":"335_CR64","doi-asserted-by":"publisher","first-page":"5929","DOI":"10.1007\/s10462-020-09838-1","volume":"53","author":"GV Houdt","year":"2020","unstructured":"Houdt GV et al (2020) A review on the long short-term memory model. Artif Intell Rev 53(8):5929\u20135955","journal-title":"Artif Intell Rev"},{"issue":"4","key":"335_CR65","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3626235","volume":"56","author":"L Yang","year":"2023","unstructured":"Yang L, Zhang Z et al (2023) Diffusion models: a comprehensive survey of methods and applications. CSUR 56(4):1\u201339","journal-title":"CSUR"},{"issue":"4","key":"335_CR66","first-page":"3313","volume":"35","author":"J Gui","year":"2023","unstructured":"Gui J, Sun Z, Wen Y et al (2023) A review on generative adversarial networks: algorithms, theory, and applications. TKDE 35(4):3313\u20133332","journal-title":"TKDE"},{"key":"335_CR67","doi-asserted-by":"crossref","unstructured":"Robertson SE, Walker S (1997) On relevance weights with little relevance information. In: SIGIR","DOI":"10.1145\/278459.258529"},{"key":"335_CR68","doi-asserted-by":"crossref","unstructured":"Lafferty JD, Zhai C (2001) Document language models, query models, and risk minimization for information retrieval. In: SIGIR","DOI":"10.1145\/383952.383970"},{"key":"335_CR69","doi-asserted-by":"crossref","unstructured":"Hershey S, Chaudhuri S et\u00a0al (2017) CNN architectures for large-scale audio classification. In: ICASSP","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"335_CR70","doi-asserted-by":"crossref","unstructured":"Dong J, Li X, Xu C et\u00a0al (2019) Dual encoding for zero-example video retrieval. In: CVPR","DOI":"10.1109\/CVPR.2019.00957"},{"key":"335_CR71","unstructured":"Xiong L, Xiong C, Li Y et\u00a0al (2021) Approximate nearest neighbor negative contrastive learning for dense text retrieval. In: ICLR"},{"issue":"9","key":"335_CR72","doi-asserted-by":"publisher","first-page":"509","DOI":"10.1145\/361002.361007","volume":"18","author":"JL Bentley","year":"1975","unstructured":"Bentley JL (1975) Multidimensional binary search trees used for associative searching. CACM 18(9):509\u2013517","journal-title":"CACM"},{"key":"335_CR73","doi-asserted-by":"crossref","unstructured":"Li W, Feng C et\u00a0al (2023) Learning balanced tree indexes for large-scale vector retrieval. In: SIGKDDg","DOI":"10.1145\/3580305.3599406"},{"key":"335_CR74","doi-asserted-by":"crossref","unstructured":"Datar M, Immorlica N, Indyk P et\u00a0al (2004) Locality-sensitive hashing scheme based on p-stable distributions. In: SCG","DOI":"10.1145\/997817.997857"},{"issue":"4","key":"335_CR75","doi-asserted-by":"publisher","first-page":"824","DOI":"10.1109\/TPAMI.2018.2889473","volume":"42","author":"YA Malkov","year":"2018","unstructured":"Malkov YA, Yashunin DA (2018) Efficient and robust approximate nearest neighbor search using hierarchical navigable small world graphs. TPAMI 42(4):824\u2013836","journal-title":"TPAMI"},{"key":"335_CR76","unstructured":"Jayaram\u00a0Subramanya S, Devvrit F et\u00a0al (2019) Diskann: fast accurate billion-point nearest neighbor search on a single node. In: NeurIPS"},{"key":"335_CR77","unstructured":"Wang Y, Hou Y et\u00a0al (2022) A neural corpus indexer for document retrieval. In: NeurIPS"},{"key":"335_CR78","doi-asserted-by":"crossref","unstructured":"Zhang H, Wang Y et\u00a0al (2023) Model-enhanced vector index. In: NeurIPS","DOI":"10.52202\/075280-2396"},{"key":"335_CR79","doi-asserted-by":"crossref","unstructured":"Hayati SA, Olivier R et\u00a0al (2018) Retrieval-based neural code generation. In: EMNLP","DOI":"10.18653\/v1\/D18-1111"},{"key":"335_CR80","doi-asserted-by":"crossref","unstructured":"Zhang J, Wang X, Zhang H et\u00a0al (2020) Retrieval-based neural source code summarization. In: ICSE","DOI":"10.1145\/3377811.3380383"},{"key":"335_CR81","unstructured":"Poesia G, Polozov A, Le V et\u00a0al (2022) Synchromesh: reliable code generation from pre-trained language models. In: ICLR"},{"key":"335_CR82","doi-asserted-by":"crossref","unstructured":"Ye X, Yavuz S et\u00a0al (2022) RNG-KBQA: generation augmented iterative ranking for knowledge base question answering. In: ACL","DOI":"10.18653\/v1\/2022.acl-long.417"},{"key":"335_CR83","doi-asserted-by":"crossref","unstructured":"Shu Y et\u00a0al (2022) TIARA: multi-grained retrieval for robust question answering over large knowledge bases. arXiv:2210.12925","DOI":"10.18653\/v1\/2022.emnlp-main.555"},{"key":"335_CR84","doi-asserted-by":"crossref","unstructured":"Lin XV, Socher R et\u00a0al (2020) Bridging textual and tabular data for cross-domain text-to-sql semantic parsing. arXiv:2012.12627","DOI":"10.18653\/v1\/2020.findings-emnlp.438"},{"key":"335_CR85","unstructured":"Asai A, Wu Z et\u00a0al (2023) Self-rag: learning to retrieve, generate, and critique through self-reflection. arxiv:2310.11511"},{"key":"335_CR86","unstructured":"Shi W, Min S et\u00a0al (2023) Replug: retrieval-augmented black-box language models. arXiv:2301.12652"},{"key":"335_CR87","doi-asserted-by":"crossref","unstructured":"Ram O, Levine Y et\u00a0al (2023) In-context retrieval-augmented language models. arXiv:2302.00083","DOI":"10.1162\/tacl_a_00605"},{"key":"335_CR88","doi-asserted-by":"crossref","unstructured":"Zan D, Chen B, Lin Z et\u00a0al (2022) When language model meets private library. In: EMNLP findings","DOI":"10.18653\/v1\/2022.findings-emnlp.21"},{"key":"335_CR89","doi-asserted-by":"crossref","unstructured":"Nashid N, Sintaha M, Mesbah A (2023) Retrieval-based prompt selection for code-related few-shot learning. In: ICSE","DOI":"10.1109\/ICSE48619.2023.00205"},{"key":"335_CR90","doi-asserted-by":"crossref","unstructured":"Jin M, Shahriar S, Tufano M et\u00a0al (2023) Inferfix: end-to-end program repair with LLMS. In: ESEC\/FSE","DOI":"10.1145\/3611643.3613892"},{"key":"335_CR91","doi-asserted-by":"crossref","unstructured":"Lu S, Duan N, Han H et\u00a0al (2022) REACC: a retrieval-augmented code completion framework. In: ACL","DOI":"10.18653\/v1\/2022.acl-long.431"},{"key":"335_CR92","doi-asserted-by":"crossref","unstructured":"Liu Y et\u00a0al (2022) Uni-parser: unified semantic parser for question answering on knowledge base and database. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.605"},{"key":"335_CR93","doi-asserted-by":"crossref","unstructured":"Yang Z, Du X, Cambria E et\u00a0al (2023) End-to-end case-based reasoning for commonsense knowledge base completion. In: EACL","DOI":"10.18653\/v1\/2023.eacl-main.255"},{"key":"335_CR94","doi-asserted-by":"crossref","unstructured":"Shi W, Zhuang Y, Zhu Y et\u00a0al (2023) Retrieval-augmented large language models for adolescent idiopathic scoliosis patients in shared decision-making. In: ACM-BCB","DOI":"10.1145\/3584371.3612956"},{"key":"335_CR95","unstructured":"Casanova A, Careil M, Verbeek J et\u00a0al (2021) Instance-conditioned GAN. In: NeurIPS"},{"key":"335_CR96","unstructured":"Bertsch A, Alon U et\u00a0al (2023) Unlimiformer: long-range transformers with unlimited length input In: NeurIPS. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/hash\/6f9806a5adc72b5b834b27e4c7c0df9b-Abstract-Conference.html"},{"key":"335_CR97","unstructured":"Kuratov Y, Bulatov A et\u00a0al (2024) In search of needles in a 10M haystack: recurrent memory finds what LLMs Miss. arXiv:2402.10790"},{"key":"335_CR98","doi-asserted-by":"crossref","unstructured":"Li J, Li Y et\u00a0al (2021) EditSum: a retrieve-and-edit framework for source code summarization. In: ASE","DOI":"10.1109\/ASE51524.2021.9678724"},{"key":"335_CR99","doi-asserted-by":"crossref","unstructured":"Yu C, Yang G, Chen X et\u00a0al (2022) BashExplainer: retrieval-augmented Bash code comment generation based on fine-tuned CodeBERT. In: ICSME","DOI":"10.1109\/ICSME55016.2022.00016"},{"key":"335_CR100","unstructured":"Hashimoto TB, Guu K et\u00a0al (2018) A retrieve-and-edit framework for predicting structured outputs. In: NeurIPS"},{"key":"335_CR101","doi-asserted-by":"crossref","unstructured":"Wei B, Li Y, Li G et\u00a0al (2020) Retrieve and refine: exemplar-based neural comment generation. In: ASE","DOI":"10.1145\/3324884.3416578"},{"key":"335_CR102","doi-asserted-by":"crossref","unstructured":"Shi E, Wang Y, Tao W et\u00a0al (2022) RACE: retrieval-augmented commit message generation. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.372"},{"key":"335_CR103","unstructured":"Chen W, Hu H, Saharia C, Cohen WW (2023) Re-imagen: retrieval-augmented text-to-image generator. In: ICLR"},{"key":"335_CR104","unstructured":"Sheynin S, Ashual O et\u00a0al (2023) KNN-diffusion: Image generation via large-scale retrieval. In: ICLR"},{"key":"335_CR105","doi-asserted-by":"crossref","unstructured":"Blattmann A, Rombach R, Oktay K et\u00a0al (2022) Retrieval-augmented diffusion models. In: NeurIPS","DOI":"10.52202\/068431-1114"},{"key":"335_CR106","doi-asserted-by":"crossref","unstructured":"Rombach R, Blattmann A, Ommer B (2022) Text-guided synthesis of artistic images with retrieval-augmented diffusion models. arXiv:2207.13038","DOI":"10.52202\/068431-1114"},{"key":"335_CR107","unstructured":"Li B, Torr PH et\u00a0al (2022) Memory-driven text-to-image generation. arXiv:2208.07022"},{"key":"335_CR108","doi-asserted-by":"crossref","unstructured":"Oguz B, Chen X, Karpukhin V et\u00a0al (2022) UniK-QA: unified representations of structured and unstructured knowledge for open-domain question answering. In: NAACL findings","DOI":"10.18653\/v1\/2022.findings-naacl.115"},{"key":"335_CR109","unstructured":"Yu D, Zhang S et\u00a0al (2023) Decaf: joint decoding of answers and logical forms for question answering over knowledge bases. In: ICLR"},{"key":"335_CR110","doi-asserted-by":"crossref","unstructured":"Dong G, Li R, Wang S et\u00a0al (2023) Bridging the KB-text gap: leveraging structured knowledge-aware pre-training for KBQA. In: CIKM","DOI":"10.1145\/3583780.3615150"},{"key":"335_CR111","unstructured":"Wang K, Duan F, Wang S et\u00a0al (2023) Knowledge-driven cot: exploring faithful reasoning in LLMs for knowledge-intensive question answering. arXiv:2308.13259"},{"key":"335_CR112","doi-asserted-by":"crossref","unstructured":"Yu D, Yang Y (2023) Retrieval-enhanced generative model for large-scale knowledge graph completion. In: SIGIR","DOI":"10.1145\/3539618.3592052"},{"key":"335_CR113","doi-asserted-by":"crossref","unstructured":"F\u00e9vry T, Soares LB et\u00a0al (2020) Entities as experts: sparse memory access with entity supervision. In: EMNLP","DOI":"10.18653\/v1\/2020.emnlp-main.400"},{"key":"335_CR114","unstructured":"Jong M, Zemlyanskiy Y et\u00a0al (2021) Mention memory: incorporating textual knowledge into transformers through entity mention attention. In: ICLR"},{"key":"335_CR115","doi-asserted-by":"crossref","unstructured":"Jing B, Zhang Y, Song Z et\u00a0al (2024) AMD: anatomical motion diffusion with interpretable motion decomposition and fusion. In: AAAI","DOI":"10.1609\/aaai.v38i3.28042"},{"key":"335_CR116","doi-asserted-by":"crossref","unstructured":"Yuan Y, Liu H, Liu X et\u00a0al (2024) Retrieval-augmented text-to-audio generation. In: ICASSP","DOI":"10.1109\/ICASSP48485.2024.10447898"},{"key":"335_CR117","first-page":"5366","volume":"32","author":"B Yang","year":"2023","unstructured":"Yang B, Cao M, Zou Y (2023) Concept-aware video captioning: describing videos with effective prior information. TIP 32:5366\u20135378","journal-title":"TIP"},{"key":"335_CR118","doi-asserted-by":"crossref","unstructured":"Zhong Z, Lei T, Chen D (2022) Training language models with memory augmentation. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.382"},{"key":"335_CR119","doi-asserted-by":"crossref","unstructured":"Min S, Shi W et\u00a0al (2023) Nonparametric masked language modeling. In: ACL findings","DOI":"10.18653\/v1\/2023.findings-acl.132"},{"key":"335_CR120","doi-asserted-by":"crossref","unstructured":"Zhang X, Zhou Y, Yang G, Chen T (2023) Syntax-aware retrieval augmented code generation. In: EMNLP findings","DOI":"10.18653\/v1\/2023.findings-emnlp.90"},{"key":"335_CR121","doi-asserted-by":"crossref","unstructured":"Fei Z (2021) Memory-augmented image captioning. In: AAAI","DOI":"10.1609\/aaai.v35i2.16220"},{"key":"335_CR122","unstructured":"Leviathan Y, Kalman M, Matias Y (2023) Fast inference from transformers via speculative decoding. In: ICML"},{"key":"335_CR123","unstructured":"Lan T, Cai D, Wang Y et\u00a0al (2023) Copy is all you need. In: ICLR"},{"key":"335_CR124","unstructured":"Cao B, Cai D, Cui L et al (2024) Retrieval is accurate generation. arXiv:2402.17532"},{"key":"335_CR125","doi-asserted-by":"crossref","unstructured":"Wang L, Yang N, Wei F (2023) Query2doc: query expansion with large language models. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.585"},{"key":"335_CR126","doi-asserted-by":"crossref","unstructured":"Gao L, Ma X, Lin J, Callan J (2023) Precise zero-shot dense retrieval without relevance labels. In: ACL","DOI":"10.18653\/v1\/2023.acl-long.99"},{"key":"335_CR127","doi-asserted-by":"crossref","unstructured":"Kim G, Kim S, Jeon B et al (2023) Tree of clarifications: answering ambiguous questions with retrieval-augmented large language models. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.63"},{"key":"335_CR128","unstructured":"Chan C-M, Xu C et al (2024) Rq-rag: learning to refine queries for retrieval augmented generation. arXiv:2404.00610"},{"key":"335_CR129","doi-asserted-by":"crossref","unstructured":"Tayal A, Tyagi A (2024) Dynamic contexts for generating suggestion questions in rag based conversational systems. In: WWW\u201924 Companion","DOI":"10.1145\/3589335.3651905"},{"key":"335_CR130","unstructured":"Xia M, Malladi S, Gururangan S et al (2024) LESS: selecting influential data for targeted instruction tuning. arXiv:2402.04333"},{"key":"335_CR131","doi-asserted-by":"crossref","unstructured":"Bornea A-L, Ayed F et al (2024) Telco-rag: navigating the challenges of retrieval-augmented language models for telecommunications. arXiv:2404.15939","DOI":"10.1109\/GLOBECOM52923.2024.10901158"},{"key":"335_CR132","unstructured":"Tan L, Huang K-W, Shi J, Wu K (2025) Interpdetect: Interpretable signals for detecting hallucinations in retrieval-augmented generation. arXiv preprint arXiv:2510.21538"},{"key":"335_CR133","unstructured":"Yao S, Zhao J et al (2023) React: synergizing reasoning and acting in language models. In: ICLR"},{"key":"335_CR134","unstructured":"Wei J, Wang X, Schuurmans D et al (2022) Chain-of-thought prompting elicits reasoning in large language models. In: NeurIPS"},{"key":"335_CR135","unstructured":"Pouplin T, Sun H, Holt S, Schaar M (2024) Retrieval-augmented thought process as sequential decision making. arXiv:2402.07812"},{"key":"335_CR136","unstructured":"Liu J. LlamaIndex. https:\/\/github.com\/jerryjliu\/llama_index"},{"key":"335_CR137","unstructured":"Sarthi P, Abdullah S, Tuli A et al (2023) Raptor: recursive abstractive processing for tree-organized retrieval. In: ICLR"},{"key":"335_CR138","unstructured":"Kang B, Kim J et al (2024) Prompt-rag: pioneering vector embedding-free retrieval-augmented generation in niche domains, exemplified by Korean medicine. arXiv:2401.11246"},{"key":"335_CR139","doi-asserted-by":"crossref","unstructured":"Raina V et al (2024) Question-based retrieval using atomic units for enterprise rag. arXiv:2405.12363","DOI":"10.18653\/v1\/2024.fever-1.25"},{"key":"335_CR140","doi-asserted-by":"crossref","unstructured":"Zhao J, Ji Z, Niu S, Wang H, Xiong F, Li Z (2025) Mom: mixtures of scenario-aware document memories for retrieval-augmented generation systems. arXiv preprint arXiv:2510.14252","DOI":"10.18653\/v1\/2025.acl-long.258"},{"key":"335_CR141","unstructured":"Xiao S, Liu Z, Zhang P et al (2023) C-pack: packaged resources to advance general Chinese embedding. arxiv:2309.07597"},{"key":"335_CR142","doi-asserted-by":"crossref","unstructured":"Chen J, Xiao S, Zhang P et al (2023) Bge m3-embedding: multi-lingual, multi-functionality, multi-granularity text embeddings through self-knowledge distillation. arxiv:2309.07597","DOI":"10.18653\/v1\/2024.findings-acl.137"},{"key":"335_CR143","doi-asserted-by":"crossref","unstructured":"Xiao S, Liu Z, Zhang P, Xing X (2023) Lm-cocktail: resilient tuning of language models via model merging. arxiv:2311.13534","DOI":"10.18653\/v1\/2024.findings-acl.145"},{"key":"335_CR144","unstructured":"Zhang P, Xiao S., Liu Z, Dou Z, Nie J-Y (2023) Retrieve anything to augment large language models. arxiv:2310.07554"},{"key":"335_CR145","unstructured":"Kulkarni M, Tangarajan P, Kim K et al (2024) Reinforcement learning for optimizing RAG for domain chatbots. arXiv:2401.06800"},{"key":"335_CR146","doi-asserted-by":"crossref","unstructured":"Wang W, Wang Y et al (2023) Rap-gen: retrieval-augmented patch generation with codet5 for automatic program repair. In: ESEC\/FSE","DOI":"10.1145\/3611643.3616256"},{"key":"335_CR147","doi-asserted-by":"crossref","unstructured":"Sawarkar K, Mangal A et al (2024) Blended rag: improving rag (retriever-augmented generation) accuracy with semantic search and hybrid query-based retrievers. arXiv:2404.07220","DOI":"10.1109\/MIPR62202.2024.00031"},{"key":"335_CR148","doi-asserted-by":"crossref","unstructured":"Yan S-Q, Gu J-C, Zhu Y, Ling Z-H (2024) Corrective retrieval augmented generation. arXiv:2401.15884","DOI":"10.2139\/ssrn.5267341"},{"key":"335_CR149","doi-asserted-by":"crossref","unstructured":"Huang W, Lapata M, Vougiouklis P et al (2023) Retrieval augmented generation with rich answer encoding. In: IJCNLP-AACL","DOI":"10.18653\/v1\/2023.ijcnlp-main.65"},{"key":"335_CR150","unstructured":"Wang H, Huang W, Deng Y et al (2024) Unims-rag: a unified multi-source retrieval-augmented generation for personalized dialogue systems. arXiv:2401.13256"},{"key":"335_CR151","doi-asserted-by":"crossref","unstructured":"Koley S, Bhunia AK et al (2024) You\u2019ll never walk alone: a sketch and text duet for fine-grained image retrieval. In: CVPR","DOI":"10.1109\/CVPR52733.2024.01562"},{"key":"335_CR152","doi-asserted-by":"crossref","unstructured":"Glass MR, Rossiello G, Chowdhury MFM et al (2022) Re2g: retrieve, rerank, generate. In: NAACL","DOI":"10.18653\/v1\/2022.naacl-main.194"},{"key":"335_CR153","unstructured":"Nogueira RF, Cho K (2019) Passage re-ranking with BERT. arxiv:1901.04085"},{"key":"335_CR154","unstructured":"Li J, Zhao Y, Li Y et al (2023) Acecoder: utilizing existing code to enhance code generation. arXiv:2303.17780"},{"key":"335_CR155","doi-asserted-by":"crossref","unstructured":"Shi P, Zhang R, Bai H, Lin J (2022) XRICL: cross-lingual retrieval-augmented in-context learning for cross-lingual text-to-SQL semantic parsing. In: EMNLP Findings","DOI":"10.18653\/v1\/2022.findings-emnlp.384"},{"key":"335_CR156","doi-asserted-by":"crossref","unstructured":"Rangan K, Yin Y (2024) A fine-tuning enhanced rag system with quantized influence measure as AI judge. arXiv:2402.17081","DOI":"10.1038\/s41598-024-79110-x"},{"key":"335_CR157","doi-asserted-by":"crossref","unstructured":"Saad-Falcon J, Khattab O, Santhanam K et al (2023) Udapdr: unsupervised domain adaptation via LLM prompting and distillation of rerankers. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.693"},{"key":"335_CR158","unstructured":"Wang L, Yang N, Wei F (2023) Learning to retrieve in-context examples for large language models. arXiv:2307.07164"},{"key":"335_CR159","unstructured":"Finardi P, Avila L et al (2024) The chronicles of rag: the retriever, the chunk and the generator. arXiv:2401.07883"},{"key":"335_CR160","unstructured":"Li J, Yuan Y, Zhang Z (2024) Enhancing LLM factual accuracy with rag to counter hallucinations: a case study on domain-specific queries in private knowledge-bases. arXiv:2403.10446"},{"key":"335_CR161","unstructured":"Yuan Y, Shabani MA, Liu S (2025) Embedding-based context-aware reranker. arXiv preprint arXiv:2510.13329"},{"key":"335_CR162","unstructured":"Wang Z, Araki J, Jiang Z et al (2023) Learning to filter context for retrieval-augmented generation. arxiv:2311.08377"},{"key":"335_CR163","doi-asserted-by":"crossref","unstructured":"Hofst\u00e4tter S, Chen J, Raman K, Zamani H (2023) Fid-light: efficient and effective retrieval-augmented text generation. In: SIGIR","DOI":"10.1145\/3539618.3591687"},{"key":"335_CR164","unstructured":"Arora D, Kini A, Chowdhury SR et al (2023) Gar-meets-rag paradigm for zero-shot information retrieval. arXiv:2310.20158"},{"key":"335_CR165","unstructured":"https:\/\/www.pinecone.io"},{"key":"335_CR166","unstructured":"Yu W, Iter D et al (2022) Generate rather than retrieve: large language models are strong context generators. arXiv:2209.10063"},{"key":"335_CR167","unstructured":"Abdallah A, Jatowt A (2023) Generator-retriever-generator: a novel approach to open-domain question answering. arXiv:2307.11278"},{"key":"335_CR168","unstructured":"Best M, Kubicek A et al (2024) Multi-head rag: solving multi-aspect problems with LLMs. arXiv:2406.05085"},{"key":"335_CR169","unstructured":"Saravia E (2022) Prompt engineering guide. https:\/\/github.com\/dair-ai\/Prompt-Engineering-Guide"},{"key":"335_CR170","unstructured":"Zheng HS, Mishra S et al (2023) Take a step back: evoking reasoning via abstraction in large language models. arxiv:2310.06117"},{"key":"335_CR171","unstructured":"Diao S, Wang P, Lin Y, Zhang T (2023) Active prompting with chain-of-thought for large language models. arxiv:2302.12246"},{"key":"335_CR172","doi-asserted-by":"crossref","unstructured":"Jiang H, Wu Q, Lin C et al (2023) Llmlingua: compressing prompts for accelerated inference of large language models. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.825"},{"key":"335_CR173","unstructured":"Liu NF, Lin K, Hewitt J et al (2023) Lost in the middle: how language models use long contexts. arxiv:2307.03172"},{"key":"335_CR174","doi-asserted-by":"crossref","unstructured":"Ahmed T, Pai KS, Devanbu P, Barr ET (2024) Automatic semantic augmentation of language model prompts (for code summarization). arXiv:2304.06815","DOI":"10.1145\/3597503.3639183"},{"key":"335_CR175","unstructured":"Xu Z, Liu Z, Liu Y et al (2024) Activerag: revealing the treasures of knowledge via active learning. arXiv:2402.13547"},{"key":"335_CR176","unstructured":"Nijkamp E, Pang B, Hayashi H et al (2022) A conversational paradigm for program synthesis. arxiv:2203.13474"},{"key":"335_CR177","unstructured":"He Y, Xia M, Chen H et al (2023) Animate-a-story: storytelling with retrieval-augmented video generation. arXiv:2307.06940"},{"key":"335_CR178","unstructured":"Hu EJ, Shen Y et al (2022) Lora: low-rank adaptation of large language models. In: ICLR"},{"key":"335_CR179","unstructured":"Liu C, \u00c7etin P, Patodia Y et al (2023) Automated code editing with search-generate-modify. arXiv:2306.06490"},{"key":"335_CR180","doi-asserted-by":"crossref","unstructured":"Joshi H, S\u00e1nchez JPC, Gulwani S et al (2023) Repair is nearly generation: multilingual program repair with LLMs. In: AAAI","DOI":"10.1609\/aaai.v37i4.25642"},{"key":"335_CR181","doi-asserted-by":"crossref","unstructured":"Jiang Z, Xu FF et al (2023) Active retrieval augmented generation. arXiv:2305.06983","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"key":"335_CR182","doi-asserted-by":"crossref","unstructured":"Mallen A, Asai A, Zhong V et al (2023) When not to trust language models: investigating effectiveness of parametric and non-parametric memories. In: ACL","DOI":"10.18653\/v1\/2023.acl-long.546"},{"key":"335_CR183","doi-asserted-by":"crossref","unstructured":"Jiang Z, Araki J, Ding H, Neubig G (2021) How can we know When language models know? On the calibration of language models for question answering. TACL. https:\/\/direct.mit.edu\/tacl\/article\/doi\/10.1162\/tacl_a_00407\/107277\/How-Can-We-Know-When-Language-Models-Know-On-the","DOI":"10.1162\/tacl_a_00407"},{"key":"335_CR184","unstructured":"Kandpal N, Deng H, Roberts A et al (2023) Large language models struggle to learn long-tail knowledge. In: ICML"},{"key":"335_CR185","unstructured":"Ren R, Wang Y, Qu Y et al (2023) Investigating the factual knowledge boundary of large language models with retrieval augmentation. arxiv:2307.11019"},{"key":"335_CR186","doi-asserted-by":"crossref","unstructured":"Wang Y, Li P, Sun M, Liu Y (2023) Self-knowledge guided retrieval augmentation for large language models. In: EMNLP findings","DOI":"10.18653\/v1\/2023.findings-emnlp.691"},{"key":"335_CR187","unstructured":"Ding H, Pang L, Wei Z et al (2024) Retrieve only when it needs: adaptive retrieval augmentation for hallucination mitigation in large language models. arXiv:2402.10612"},{"key":"335_CR188","doi-asserted-by":"crossref","unstructured":"Jeong S, Baek J, Cho S et al (2024) Adaptive-rag: learning to adapt retrieval-augmented large language models through question complexity. arXiv:2403.14403","DOI":"10.18653\/v1\/2024.naacl-long.389"},{"key":"335_CR189","doi-asserted-by":"crossref","unstructured":"Zhang F, Chen B et al (2023) Repocoder: repository-level code completion through iterative retrieval and generation. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.151"},{"key":"335_CR190","doi-asserted-by":"crossref","unstructured":"Shao Z, Gong Y, Shen Y et al (2023) Enhancing retrieval-augmented large language models with iterative retrieval-generation synergy. In: EMNLP findings","DOI":"10.18653\/v1\/2023.findings-emnlp.620"},{"key":"335_CR191","doi-asserted-by":"crossref","unstructured":"Cheng X, Luo D, Chen X et al (2023) Lift yourself up: retrieval-augmented text generation with self-memory. In: NeurIPS","DOI":"10.52202\/075280-1899"},{"key":"335_CR192","unstructured":"Wang Z, Liu A, Lin H et al (2024) Rat: retrieval augmented thoughts elicit context-aware reasoning in long-horizon generation. arXiv:2403.05313"},{"key":"335_CR193","doi-asserted-by":"crossref","unstructured":"Agarwal O, Ge H, Shakeri S, Al-Rfou R (2021) Knowledge graph based synthetic corpus generation for knowledge-enhanced language model pre-training. In: NAACL-HLT","DOI":"10.18653\/v1\/2021.naacl-main.278"},{"key":"335_CR194","unstructured":"Sun J, Xu C et al (2023) Think-on-graph: deep and responsible reasoning of large language model with knowledge graph. arXiv:2307.07697"},{"key":"335_CR195","doi-asserted-by":"crossref","unstructured":"Limkonchotiwat P, Ponwitayarat W et al (2022) Cl-relkt: cross-lingual language knowledge transfer for multilingual retrieval question answering. In: NAACL Findings","DOI":"10.18653\/v1\/2022.findings-naacl.165"},{"key":"335_CR196","unstructured":"Asai A, Yu X et al (2021) One question answering model for many languages with cross-lingual dense passage retrieval. In: NeurIPS"},{"key":"335_CR197","doi-asserted-by":"crossref","unstructured":"Lee K, Han S et al (2023) When to read documents or QA history: on unified and selective open-domain QA. In: ACL Findings","DOI":"10.18653\/v1\/2023.findings-acl.401"},{"key":"335_CR198","unstructured":"Yue S, Chen W et al (2023) Disc-lawllm: fine-tuning large language models for intelligent legal services. arXiv:2309.11325"},{"key":"335_CR199","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/tacl_a_00530","volume":"11","author":"S Siriwardhana","year":"2023","unstructured":"Siriwardhana S, Weerasekera R, Kaluarachchi T et al (2023) Improving the domain adaptation of retrieval augmented generation (RAG) models for open domain question answering. TACL 11:1\u201317","journal-title":"TACL"},{"key":"335_CR200","unstructured":"Tang Y, Yang Y (2024) Multihop-rag: benchmarking retrieval-augmented generation for multi-hop queries. arXiv:2401.15391"},{"key":"335_CR201","unstructured":"Huang K, Zhai C, Ji H (2022) CONCRETE: improving cross-lingual fact-checking with cross-lingual retrieval. In: COLING"},{"key":"335_CR202","doi-asserted-by":"crossref","unstructured":"Hagstr\u00f6m L, Saynova D, Norlund T et al (2023) The effect of scaling, retrieval augmentation and form on the factual consistency of language models. arXiv:2311.01307","DOI":"10.18653\/v1\/2023.emnlp-main.332"},{"key":"335_CR203","doi-asserted-by":"crossref","unstructured":"Zamani H, Bendersky M (2024) Stochastic rag: end-to-end retrieval-augmented generation through expected utility maximization. arXiv:2405.02816","DOI":"10.1145\/3626772.3657923"},{"key":"335_CR204","doi-asserted-by":"crossref","unstructured":"Liu Y, Wan Y et al (2021) KG-BART: knowledge graph-augmented BART for generative commonsense reasoning. In: AAAI","DOI":"10.1609\/aaai.v35i7.16796"},{"key":"335_CR205","doi-asserted-by":"crossref","unstructured":"Wan A, Wallace E, Klein D (2024) What evidence do language models find convincing? arXiv:2402.11782","DOI":"10.18653\/v1\/2024.acl-long.403"},{"key":"335_CR206","doi-asserted-by":"crossref","unstructured":"Zhang H, Liu Z et al (2020) Grounded conversation generation as guided traverses in commonsense knowledge graphs. In: ACL","DOI":"10.18653\/v1\/2020.acl-main.184"},{"key":"335_CR207","doi-asserted-by":"crossref","unstructured":"Cai D, Wang Y et al (2019) Skeleton-to-response: dialogue generation guided by retrieval memory. In: NAACL-HLT","DOI":"10.18653\/v1\/N19-1124"},{"key":"335_CR208","doi-asserted-by":"crossref","unstructured":"Komeili M, Shuster K, Weston J (2022) Internet-augmented dialogue generation. In: ACL","DOI":"10.18653\/v1\/2022.acl-long.579"},{"key":"335_CR209","unstructured":"Shuster K, Xu J et al (2022) Blenderbot 3: a deployed conversational agent that continually learns to responsibly engage. arXiv:2208.03188"},{"key":"335_CR210","doi-asserted-by":"crossref","unstructured":"Kim S, Jang JY et al (2021) A model of cross-lingual knowledge-grounded response generation for open-domain dialogue systems. In: EMNLP Findings","DOI":"10.18653\/v1\/2021.findings-emnlp.33"},{"key":"335_CR211","doi-asserted-by":"crossref","unstructured":"Nie E, Liang S, Schmid H, Sch\u00fctze H (2023) Cross-lingual retrieval augmented prompt for low-resource languages. In: ACL","DOI":"10.18653\/v1\/2023.findings-acl.528"},{"key":"335_CR212","unstructured":"Li X, Nie E, Liang S (2023) From classification to generation: insights into crosslingual retrieval augmented ICL. In: NeurIPS"},{"key":"335_CR213","doi-asserted-by":"crossref","unstructured":"Li W, Li J, Ma W, Liu Y (2024) Citation-enhanced generation for LLM-based chatbot. arXiv:2402.16063","DOI":"10.18653\/v1\/2024.acl-long.79"},{"key":"335_CR214","unstructured":"Cai D, Wang Y et al (2021) Neural machine translation with monolingual translation memory. In: ACL\/IJCNLP"},{"key":"335_CR215","unstructured":"Khandelwal U, Fan A et al (2021) Nearest neighbor machine translation. In: ICLR"},{"key":"335_CR216","doi-asserted-by":"crossref","unstructured":"Du X, Ji H (2022) Retrieval-augmented generative question answering for event argument extraction. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.307"},{"key":"335_CR217","doi-asserted-by":"crossref","unstructured":"Gao Y, Yin Q et al (2022) Retrieval-augmented multilingual keyphrase generation with retriever-generator iterative training. In: NAACL findings","DOI":"10.18653\/v1\/2022.findings-naacl.92"},{"key":"335_CR218","unstructured":"Zhang J, Yu EJ, Chen Q et al (2024) Retrieval-based full-length wikipedia generation for emergent events. arXiv:2402.18264"},{"key":"335_CR219","unstructured":"Fan R, Fan Y, Chen J et al (2023) RIGHT: retrieval-augmented generation for mainstream hashtag recommendation. arxiv:2312.10466"},{"key":"335_CR220","doi-asserted-by":"crossref","unstructured":"Wang Z, Teo SX et al (2024) M-rag: reinforcing large language model performance through retrieval-augmented generation with multiple partitions. arXiv:2405.16420","DOI":"10.18653\/v1\/2024.acl-long.108"},{"key":"335_CR221","unstructured":"Wang Y, Le H, Gotmare AD et al (2022) Codet5mix: a pretrained mixture of encoder-decoder transformers for code understanding and generation. https:\/\/openreview.net\/forum?id=VPCi3STZcaO"},{"key":"335_CR222","doi-asserted-by":"crossref","unstructured":"Madaan A, Zhou S et al (2022) Language models of code are few-shot commonsense learners. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.90"},{"key":"335_CR223","doi-asserted-by":"crossref","unstructured":"Wang Y, Le H, Gotmare A et al (2023) Codet5+: open code large language models for code understanding and generation. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.68"},{"key":"335_CR224","doi-asserted-by":"crossref","unstructured":"Chen J, Hu X, Li Z et al (2024) Code search is all you need? improving code suggestions with code search. In: ICSE","DOI":"10.1145\/3597503.3639085"},{"key":"335_CR225","unstructured":"Zan D, Chen B, Gong Y et al (2023) Private-library-oriented code generation with large language models. arXiv:2307.15370"},{"key":"335_CR226","doi-asserted-by":"crossref","unstructured":"Liu M, Yang T, Lou Y et al (2023) Codegen4libs: a two-stage approach for library-oriented code generation. In: ASE","DOI":"10.1109\/ASE56229.2023.00159"},{"key":"335_CR227","unstructured":"Liao D, Pan S, Huang Q et al (2023) Context-aware code generation framework for code repositories: local, global, and third-party library awareness. arXiv:2312.05772"},{"key":"335_CR228","doi-asserted-by":"crossref","unstructured":"Li J, Li Y, Li G et al (2023) Skcoder: a sketch-based approach for automatic code generation. In: ICSE","DOI":"10.1109\/ICSE48619.2023.00179"},{"key":"335_CR229","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2024.111982","volume":"211","author":"Q Gou","year":"2024","unstructured":"Gou Q, Dong Y, Wu Y, Ke Q (2024) Rrgcode: deep hierarchical search-based code generation. J Syst Softw 211:111982","journal-title":"J Syst Softw"},{"key":"335_CR230","doi-asserted-by":"crossref","unstructured":"Zhang K, Li J, Li G et al (2024) Codeagent: enhancing code generation with tool-integrated agent systems for real-world repo-level coding challenges. arXiv:2401.07339","DOI":"10.18653\/v1\/2024.acl-long.737"},{"key":"335_CR231","unstructured":"Su H, Jiang S, Lai Y et al (2024) ARKS: active retrieval in knowledge soup for code generation. arXiv:2402.12317"},{"key":"335_CR232","unstructured":"Zhang K, Li G, Li J et al (2023) Toolcoder: teach code generation models to use API search tools. arXiv:2305.04032"},{"key":"335_CR233","unstructured":"Liu S, Chen Y, Xie X et al (2021) Retrieval-augmented generation for code summarization via hybrid GNN. In: ICLR"},{"key":"335_CR234","doi-asserted-by":"crossref","unstructured":"Yamaguchi F, Golde N, Arp D, Rieck K (2014) Modeling and discovering vulnerabilities with code property graphs. In: S &P","DOI":"10.1109\/SP.2014.44"},{"key":"335_CR235","doi-asserted-by":"crossref","unstructured":"Choi Y, Na C et al (2023) Readsum: retrieval-augmented adaptive transformer for source code summarization. IEEE Access. https:\/\/ieeexplore.ieee.org\/abstract\/document\/10113620\/","DOI":"10.1109\/ACCESS.2023.3271992"},{"key":"335_CR236","volume":"168","author":"J Zhao","year":"2024","unstructured":"Zhao J, Chen X, Yang G, Shen Y (2024) Automatic smart contract comment generation via large language models and in-context learning. IST 168:107405","journal-title":"IST"},{"issue":"4","key":"335_CR237","doi-asserted-by":"publisher","first-page":"604","DOI":"10.3390\/math10040604","volume":"10","author":"A Alokla","year":"2022","unstructured":"Alokla A, Gad W, Nazih W et al (2022) Retrieval-based transformer pseudocode generation. Mathematics 10(4):604","journal-title":"Mathematics"},{"key":"335_CR238","doi-asserted-by":"crossref","unstructured":"Xu J, Cui Z et al (2024) Unilog: automatic logging via LLM and in-context learning. In: ICSE","DOI":"10.1145\/3597503.3623326"},{"issue":"4","key":"335_CR239","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1145\/3464689","volume":"30","author":"H Wang","year":"2021","unstructured":"Wang H, Xia X et al (2021) Context-aware retrieval-based deep commit message generation. TOSEM 30(4):56\u201315630","journal-title":"TOSEM"},{"key":"335_CR240","doi-asserted-by":"crossref","unstructured":"Zhu X, Sha C, Niu J (2022) A simple retrieval-based method for code comment generation. In: SANER","DOI":"10.1109\/SANER53432.2022.00126"},{"key":"335_CR241","doi-asserted-by":"crossref","unstructured":"Ye T, Wu L, Ma T et al (2023) Tram: a token-level retrieval-augmented mechanism for source code summarization. arXiv:2305.11074","DOI":"10.18653\/v1\/2024.findings-naacl.186"},{"key":"335_CR242","doi-asserted-by":"crossref","unstructured":"Li L, Liang B, Chen L, Zhang X (2024) Cross-modal retrieval-enhanced code summarization based on joint learning for retrieval and generation. Available at SSRN 4724884","DOI":"10.2139\/ssrn.4724884"},{"key":"335_CR243","unstructured":"Drain D, Hu C, Wu C et al (2021) Generating code with the help of retrieved template functions and stack overflow answers. arXiv:2104.05310"},{"key":"335_CR244","unstructured":"Eghbali A, Pradel M (2024) De-hallucinator: iterative grounding for LLM-based code completion. arXiv:2401.01701"},{"key":"335_CR245","unstructured":"Liang M, Xie X, Zhang G et al (2024) Repofuse: repository-level code completion with fused dual context. arXiv:2402.14323"},{"key":"335_CR246","unstructured":"Shrivastava D, Kocetkov D et al (2023) Repofusion: training code models to understand your repository. arXiv:2306.10998"},{"key":"335_CR247","doi-asserted-by":"crossref","unstructured":"Sun W, Li H, Yan M et al (2023) Revisiting and improving retrieval-augmented deep assertion generation. In: ASE","DOI":"10.1109\/ASE56229.2023.00090"},{"key":"335_CR248","unstructured":"Ding Y, Wang Z et al (2022) Cocomic: code completion by jointly modeling in-file and cross-file context. arXiv:2212.10007"},{"key":"335_CR249","doi-asserted-by":"crossref","unstructured":"Tang Z, Ge J, Liu S et al (2023) Domain adaptive code completion via language models and decoupled domain databases. In: ASE","DOI":"10.1109\/ASE56229.2023.00076"},{"key":"335_CR250","doi-asserted-by":"crossref","unstructured":"Tsai Y, Liu M, Ren H (2023) Rtlfixer: automatically fixing RTL syntax errors with large language models. arXiv:2311.16543","DOI":"10.1145\/3649329.3657353"},{"key":"335_CR251","unstructured":"Bogin B, Gupta S, Clark P et al (2023) Leveraging code to improve in-context learning for semantic parsing. arXiv:2311.09519"},{"key":"335_CR252","doi-asserted-by":"crossref","unstructured":"Li H, Zhang J, Li C, Chen H (2023) Resdsql: decoupling schema linking and skeleton parsing for text-to-SQL. In: AAAI","DOI":"10.1609\/aaai.v37i11.26535"},{"key":"335_CR253","doi-asserted-by":"crossref","unstructured":"Zhang K, Lin X, Wang Y et al (2023) Refsql: a retrieval-augmentation framework for text-to-SQL generation. In: EMNLP findings","DOI":"10.18653\/v1\/2023.findings-emnlp.48"},{"key":"335_CR254","doi-asserted-by":"crossref","unstructured":"Chang S, Fosler-Lussier E (2023) Selective demonstrations for cross-domain text-to-SQL. arXiv:2310.06302","DOI":"10.18653\/v1\/2023.findings-emnlp.944"},{"key":"335_CR255","doi-asserted-by":"crossref","unstructured":"Nan L, Zhao Y, Zou W et al (2023) Enhancing text-to-SQL capabilities of large language models: a study on prompt design strategies. In: EMNLP Findings","DOI":"10.18653\/v1\/2023.findings-emnlp.996"},{"key":"335_CR256","unstructured":"Zhang X, Wang D, Dou L et al (2024) Multi-hop table retrieval for open-domain text-to-SQL. arXiv:2402.10666"},{"key":"335_CR257","doi-asserted-by":"crossref","unstructured":"Li H, Zhang J, Liu H et al (2024) Codes: towards building open-source language models for text-to-SQL. arXiv:2402.16347","DOI":"10.1145\/3654930"},{"key":"335_CR258","doi-asserted-by":"crossref","unstructured":"Jie Z, Lu W (2023) Leveraging training data in few-shot prompting for numerical reasoning. arXiv:2305.18170","DOI":"10.18653\/v1\/2023.findings-acl.668"},{"key":"335_CR259","unstructured":"Gao M, Li J, Fei H et al (2023) De-fine: decomposing and refining visual programs with auto-feedback. arXiv:2311.12890"},{"key":"335_CR260","unstructured":"Hao Y, Chen W, Zhou Z, Cui W (2023) E &v: prompting large language models to perform static analysis by pseudo-code execution and verification. arXiv:2312.08477"},{"key":"335_CR261","doi-asserted-by":"crossref","unstructured":"Guo Y, Li Z et al (2023) Retrieval-augmented code generation for universal information extraction. arXiv:2311.02962","DOI":"10.1007\/978-981-97-9434-8_3"},{"key":"335_CR262","doi-asserted-by":"crossref","unstructured":"Pinto G, Souza C et al (2024) Lessons from building stackspot AI: acontextualized ai coding assistant. arXiv:2311.18450","DOI":"10.1145\/3639477.3639751"},{"key":"335_CR263","doi-asserted-by":"crossref","unstructured":"Liu Z, Chen C, Wang J et al (2023) Testing the limits: unusual text inputs generation for mobile app crash detection with large language model. arXiv:2310.15657","DOI":"10.1145\/3597503.3639118"},{"key":"335_CR264","doi-asserted-by":"crossref","unstructured":"Bollacker KD, Evans C et al (2008) Freebase: a collaboratively created graph database for structuring human knowledge. In: SIGMOD","DOI":"10.1145\/1376616.1376746"},{"key":"335_CR265","unstructured":"Patidar M, Singh AK, Sawhney R et al (2023) Combining transfer learning with in-context learning using blackbox LLMs for zero-shot knowledge base question answering. arXiv:2311.08894"},{"key":"335_CR266","unstructured":"Shu Y, Yu Z (2023) Data distribution bottlenecks in grounding language models to knowledge bases. arXiv:2309.08345"},{"key":"335_CR267","doi-asserted-by":"crossref","unstructured":"Leake D, Crandall DJ (2020) On bringing case-based reasoning methodology to deep learning. In: ICCBR","DOI":"10.1007\/978-3-030-58342-2_22"},{"key":"335_CR268","doi-asserted-by":"crossref","unstructured":"Zhang L, Zhang J et al (2023) FC-KBQA: a fine-to-coarse composition framework for knowledge base question answering. In: ACL","DOI":"10.18653\/v1\/2023.acl-long.57"},{"key":"335_CR269","doi-asserted-by":"crossref","unstructured":"Jiang J, Zhou K et al (2023) Structgpt: a general framework for large language model to reason over structured data. In: EMNLP","DOI":"10.18653\/v1\/2023.emnlp-main.574"},{"key":"335_CR270","doi-asserted-by":"crossref","unstructured":"Baek J, Aji AF, Saffari A (2023) Knowledge-augmented language model prompting for zero-shot knowledge graph question answering. arXiv:2306.04136","DOI":"10.18653\/v1\/2023.nlrse-1.7"},{"key":"335_CR271","doi-asserted-by":"crossref","unstructured":"Sen P, Mavadia S, Saffari A (2023) Knowledge graph-augmented language models for complex question answering. In: NLRSE","DOI":"10.18653\/v1\/2023.nlrse-1.1"},{"key":"335_CR272","unstructured":"Wu Y, Hu N, Bi S et al (2023) Retrieve-rewrite-answer: a kg-to-text enhanced LLMs framework for knowledge graph question answering. arXiv:2309.11206"},{"key":"335_CR273","unstructured":"Wang C, Xu Y, Peng Z et al (2024) keqing: knowledge-based question answering is a nature chain-of-thought mentor of LLM. arXiv:2401.00426"},{"key":"335_CR274","unstructured":"Liu J, Cao S, Shi J et al (2024) Probing structured semantics understanding and generation of language models via question answering. arXiv:2401.05777"},{"key":"335_CR275","doi-asserted-by":"crossref","unstructured":"Xiong G, Bao J, Zhao W (2024) Interactive-kbqa: multi-turn interactions for knowledge base question answering with large language models. arXiv:2402.15131","DOI":"10.18653\/v1\/2024.acl-long.569"},{"key":"335_CR276","doi-asserted-by":"crossref","unstructured":"Chen S, Liu Q, Yu Z et al (2021) Retrack: a flexible and efficient framework for knowledge base question answering. In: ACL","DOI":"10.18653\/v1\/2021.acl-demo.39"},{"key":"335_CR277","doi-asserted-by":"crossref","unstructured":"Yu D, Zhu C, Fang Y et al (2022) Kg-fid: infusing knowledge graph in fusion-in-decoder for open-domain question answering. In: ACL","DOI":"10.18653\/v1\/2022.acl-long.340"},{"key":"335_CR278","doi-asserted-by":"crossref","unstructured":"Ju M, Yu W, Zhao T et al (2022) Grape: knowledge graph enhanced passage reader for open-domain question answering. In: EMNLP Findings","DOI":"10.18653\/v1\/2022.findings-emnlp.13"},{"key":"335_CR279","doi-asserted-by":"crossref","unstructured":"Hu Z, Xu Y, Yu W et al (2022) Empowering language models with knowledge graph reasoning for open-domain question answering. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.650"},{"key":"335_CR280","doi-asserted-by":"crossref","unstructured":"Yang Q, Chen Q, Wang W et al (2023) Enhancing multi-modal multi-hop question answering via structured knowledge and unified retrieval-generation. In: MM","DOI":"10.1145\/3581783.3611964"},{"key":"335_CR281","doi-asserted-by":"crossref","unstructured":"Zhao W, Liu Y, Niu T et al (2023) DIVKNOWQA: assessing the reasoning ability of LLMs via open-domain question answering over knowledge base and text. arXiv:2310.20170","DOI":"10.18653\/v1\/2024.findings-naacl.5"},{"key":"335_CR282","unstructured":"Wang X, Yang Q, Qiu Y et al (2023) Knowledgpt: enhancing large language models with retrieval and storage access on knowledge bases. arXiv:2308.11761"},{"key":"335_CR283","doi-asserted-by":"crossref","unstructured":"Ko S, Cho H, Chae H et al (2024) Evidence-focused fact summarization for knowledge-augmented zero-shot question answering. arXiv:2403.02966","DOI":"10.18653\/v1\/2024.emnlp-main.594"},{"key":"335_CR284","doi-asserted-by":"crossref","unstructured":"Gao Y, Qiao L, Kan Z et al (2024) Two-stage generative question answering on temporal knowledge graph using large language models. arXiv:2402.16568","DOI":"10.18653\/v1\/2024.findings-acl.401"},{"key":"335_CR285","unstructured":"Guo T, Yang Q, Wang C et al (2023) Knowledgenavigator: leveraging large language models for enhanced reasoning over knowledge graph. arXiv:2312.15880"},{"key":"335_CR286","doi-asserted-by":"crossref","unstructured":"Mavromatis C, Karypis G (2024) Gnn-rag: graph neural retrieval for large language model reasoning. arXiv:2405.20139","DOI":"10.18653\/v1\/2025.findings-acl.856"},{"key":"335_CR287","unstructured":"Min S, Boyd-Graber J, Alberti C et al (2021) Neurips 2020 efficientqa competition: systems, analyses and lessons learned. In: NeurIPS 2020 Competition and Demonstration Track"},{"key":"335_CR288","doi-asserted-by":"crossref","unstructured":"Li AH, Ng P, Xu P et al (2021) Dual reader-parser on hybrid textual and tabular evidence for open domain question answering. In: ACL\/IJCNLP","DOI":"10.18653\/v1\/2021.acl-long.315"},{"key":"335_CR289","doi-asserted-by":"crossref","unstructured":"Ma K, Cheng H, Liu X et al (2022) Open-domain question answering via chain of reasoning over heterogeneous knowledge. In: EMNLP Findings","DOI":"10.18653\/v1\/2022.findings-emnlp.392"},{"key":"335_CR290","doi-asserted-by":"crossref","unstructured":"Christmann P, Roy RS, Weikum G (2022) Conversational question answering on heterogeneous sources. In: SIGIR","DOI":"10.1145\/3477495.3531815"},{"key":"335_CR291","doi-asserted-by":"crossref","unstructured":"Park E, Lee S-M et al (2023) Rink: reader-inherited evidence reranker for table-and-text open domain question answering. In: AAAI","DOI":"10.1609\/aaai.v37i11.26577"},{"key":"335_CR292","doi-asserted-by":"crossref","unstructured":"Zhao W, Liu Y, Wan Y et al (2023) Localize, retrieve and fuse: a generalized framework for free-form question answering over tables. arXiv:2309.11049","DOI":"10.18653\/v1\/2023.findings-ijcnlp.1"},{"key":"335_CR293","unstructured":"Pan F, Canim M et al (2022) End-to-end table question answering via retrieval-augmented generation. arXiv:2203.16714"},{"key":"335_CR294","doi-asserted-by":"crossref","unstructured":"Jiang Z, Mao Y, He P et al (2022) Omnitab: pretraining with natural and synthetic data for few-shot table-based question answering. In: NAACL","DOI":"10.18653\/v1\/2022.naacl-main.68"},{"key":"335_CR295","doi-asserted-by":"crossref","unstructured":"Zhong W, Huang J, Liu Q et al (2022) Reasoning over hybrid chain for table-and-text open domain question answering. In: IJCAI","DOI":"10.24963\/ijcai.2022\/629"},{"key":"335_CR296","doi-asserted-by":"crossref","unstructured":"Sundar AS, Heck L (2023) ctbl: augmenting large language models for conversational tables. arXiv:2303.12024","DOI":"10.18653\/v1\/2023.nlp4convai-1.6"},{"key":"335_CR297","doi-asserted-by":"crossref","unstructured":"Min D, Hu N, Jin R et al (2024) Exploring the impact of table-to-text methods on augmenting LLM-based question answering with domain hybrid data. arXiv:2402.12869","DOI":"10.18653\/v1\/2024.naacl-industry.41"},{"key":"335_CR298","doi-asserted-by":"crossref","unstructured":"Roychowdhury S, Krema M et al (2024) Eratta: extreme rag for table to answers with large language models. arXiv:2405.03963","DOI":"10.1109\/BigData62323.2024.10825910"},{"key":"335_CR299","doi-asserted-by":"publisher","first-page":"74899","DOI":"10.52202\/079017-2382","volume":"37","author":"S-A Chen","year":"2024","unstructured":"Chen S-A, Miculicich L, Eisenschlos J, Wang Z, Wang Z, Chen Y, Fujii Y, Lin H-T, Lee C-Y, Pfister T (2024) Tablerag: million-token table understanding with language models. Adv Neural Inf Process Syst 37:74899\u201374921","journal-title":"Adv Neural Inf Process Syst"},{"key":"335_CR300","unstructured":"Kim K, Kim M, Lee H, Park S, Han Y, Jeon B-K (2024) Thorr: complex table retrieval and refinement for rag. In: Proceedings of the Workshop Information Retrieval\u2019s Role in RAG Systems (IR-RAG 2024) Co-located with the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, vol 3784, pp 50\u201355"},{"key":"335_CR301","doi-asserted-by":"crossref","unstructured":"Wu S, Li Y, Zhang D, Wu Z (2020) Improving knowledge-aware dialogue response generation by using human-written prototype dialogues. In: EMNLP Findings","DOI":"10.18653\/v1\/2020.findings-emnlp.126"},{"key":"335_CR302","unstructured":"Kang M, Kwak JM et al (2022) Knowledge-consistent dialogue generation with knowledge graphs. In: ICML Workshop"},{"key":"335_CR303","doi-asserted-by":"crossref","unstructured":"Ji Z, Liu Z, Lee N et al (2023) RHO: reducing hallucination in open-domain dialogues with knowledge grounding. In: ACL findings","DOI":"10.18653\/v1\/2023.findings-acl.275"},{"key":"335_CR304","doi-asserted-by":"crossref","unstructured":"Baek J, Chandrasekaran N, Cucerzan S et al (2023) Knowledge-augmented large language models for personalized contextual query suggestion. arXiv:2311.06318","DOI":"10.1145\/3589334.3645404"},{"key":"335_CR305","unstructured":"He X, Tian Y, Sun Y et al (2024) G-retriever: retrieval-augmented generation for textual graph understanding and question answering. arXiv:2402.07630"},{"key":"335_CR306","doi-asserted-by":"crossref","unstructured":"Hussien MM, Melo AN et al (2024) Rag-based explainable prediction of road users behaviors for automated driving using knowledge graphs and large language models. arXiv:2405.00449","DOI":"10.1016\/j.eswa.2024.125914"},{"key":"335_CR307","doi-asserted-by":"crossref","unstructured":"Guti\u00e9rrez BJ, Shu Y et al (2024) Hipporag: neurobiologically inspired long-term memory for large language models. arXiv:2405.14831","DOI":"10.52202\/079017-1902"},{"key":"335_CR308","unstructured":"Sourati Z, Wang Z, Liu MM, Hu Y, Guo M, Bharadwaj S, Han K, Sheng T, Ravi S, Dehghani M et al (2025) Lad-rag: layout-aware dynamic rag for visually-rich document understanding. arXiv preprint arXiv:2510.07233"},{"key":"335_CR309","unstructured":"Kirstain Y, Levy O, Polyak A (2023) X &fuse: fusing visual information in text-to-image generation. arXiv:2303.01000"},{"key":"335_CR310","unstructured":"Dhariwal P, Nichol A (2021) Diffusion models beat GANs on image synthesis. In: NeurIPS"},{"key":"335_CR311","unstructured":"Yang L, Yu Z, Meng C et al (2024) Mastering text-to-image diffusion: recaptioning, planning, and generating with multimodal LLMs. arXiv:2401.11708"},{"key":"335_CR312","unstructured":"Zhang Z, Zhang A, Li M et al (2023) Multimodal chain-of-thought reasoning in language models. arXiv:2302.00923"},{"key":"335_CR313","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106730","volume":"214","author":"C Xu","year":"2021","unstructured":"Xu C, Yang M, Ao X et al (2021) Retrieval-enhanced adversarial training with dynamic memory-augmented attention for image paragraph captioning. Knowl-Based Syst 214:106730","journal-title":"Knowl-Based Syst"},{"key":"335_CR314","doi-asserted-by":"crossref","unstructured":"Ramos R, Elliott D, Martins B (2023) Retrieval-augmented image captioning. In: EACL","DOI":"10.18653\/v1\/2023.eacl-main.266"},{"key":"335_CR315","doi-asserted-by":"crossref","unstructured":"Hu Z, Iscen A, Sun C et al (2023) Reveal: retrieval-augmented visual-language pre-training with multi-source multimodal knowledge memory. In: CVPR","DOI":"10.1109\/CVPR52729.2023.02238"},{"issue":"1","key":"335_CR316","doi-asserted-by":"publisher","first-page":"196","DOI":"10.3390\/rs16010196","volume":"16","author":"Z Li","year":"2024","unstructured":"Li Z, Zhao W, Du X et al (2024) Cross-modal retrieval and semantic refinement for remote sensing image captioning. Remote Sens 16(1):196","journal-title":"Remote Sens"},{"key":"335_CR317","doi-asserted-by":"crossref","unstructured":"Yang Z, Gan Z, Wang J et al (2022) An empirical study of GPT-3 for few-shot knowledge-based VQA. In: AAAI","DOI":"10.1609\/aaai.v36i3.20215"},{"key":"335_CR318","doi-asserted-by":"crossref","unstructured":"Lin W, Byrne B (2022) Retrieval augmented visual question answering with outside knowledge. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.772"},{"key":"335_CR319","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1162\/tacl_a_00356","volume":"9","author":"A Fan","year":"2021","unstructured":"Fan A, Gardent C, Braud C, Bordes A (2021) Augmenting transformers with KNN-based composite memory for dialog. TACL 9:82\u201399","journal-title":"TACL"},{"key":"335_CR320","unstructured":"Liang Z, Hu H, Xu C et al (2021) Maria: a visual experience powered conversational agent. In: ACL-IJCNLP"},{"key":"335_CR321","doi-asserted-by":"crossref","unstructured":"Fang Q, Feng Y (2022) Neural machine translation with phrase-level universal visual representations. In: ACL","DOI":"10.18653\/v1\/2022.acl-long.390"},{"key":"335_CR322","doi-asserted-by":"crossref","unstructured":"Whitehead S, Ji H, Bansal M et al (2018) Incorporating background knowledge into video description generation. In: EMNLP","DOI":"10.18653\/v1\/D18-1433"},{"issue":"9","key":"335_CR323","first-page":"3159","volume":"31","author":"C Yin","year":"2019","unstructured":"Yin C, Tang J, Xu Z, Wang Y (2019) Memory augmented deep recurrent neural network for video question answering. TNNLS 31(9):3159\u20133167","journal-title":"TNNLS"},{"key":"335_CR324","doi-asserted-by":"crossref","unstructured":"Pan J, Lin Z, Ge Y et al (2023) Retrieving-to-answer: zero-shot video question answering with frozen large language models. In: ICCV","DOI":"10.1109\/ICCVW60793.2023.00035"},{"key":"335_CR325","doi-asserted-by":"crossref","unstructured":"Lei J, Yu L, Berg TL, Bansal M (2020) Tvqa+: spatio-temporal grounding for video question answering. In: ACL","DOI":"10.18653\/v1\/2020.acl-main.730"},{"key":"335_CR326","doi-asserted-by":"crossref","unstructured":"Le H, Chen N, Hoi S (2022) Vgnmn: video-grounded neural module networks for video-grounded dialogue systems. In: NAACL","DOI":"10.18653\/v1\/2022.naacl-main.247"},{"key":"335_CR327","unstructured":"Wang Z, Li M, Xu R et al (2022) Language models with image descriptors are strong few-shot video-language learners. In: NeurIPS"},{"key":"335_CR328","doi-asserted-by":"crossref","unstructured":"Yuan J, Sun S, Omeiza D et al (2024) Rag-driver: generalisable driving explanations with retrieval-augmented in-context learning in multi-modal large language model. arXiv:2402.10828","DOI":"10.15607\/RSS.2024.XX.075"},{"key":"335_CR329","doi-asserted-by":"crossref","unstructured":"Ghosh S, Kumar S, Evuru CKR et al (2024) Recap: retrieval-augmented audio captioning. In: ICASSP","DOI":"10.1109\/ICASSP48485.2024.10448030"},{"key":"335_CR330","doi-asserted-by":"crossref","unstructured":"Elizalde B, Deshmukh S, Wang H (2024) Natural language supervision for general-purpose audio representations. In: ICASSP","DOI":"10.1109\/ICASSP48485.2024.10448504"},{"key":"335_CR331","unstructured":"Kouzelis T, Katsouros V (2023) Weakly-supervised automated audio captioning via text only training. In: DCASE Workshop"},{"key":"335_CR332","doi-asserted-by":"crossref","unstructured":"Deshmukh S, Elizalde B, Emmanouilidou D et al (2024) Training audio captioning models without audio. In: ICASSP","DOI":"10.1109\/ICASSP48485.2024.10448115"},{"key":"335_CR333","unstructured":"Wang Z, Lu C, Wang Y et al (2024) Prolificdreamer: high-fidelity and diverse text-to-3d generation with variational score distillation. In: NeurIPS"},{"key":"335_CR334","unstructured":"Jia H, Zhao P, Wu H, Gao Y, Tao Y, Cui B (2025) Learning from history: a retrieval-augmented framework for spatiotemporal prediction. arXiv preprint arXiv:2510.24049"},{"key":"335_CR335","unstructured":"Ning K, Pan Z, Liu Y, Jiang Y, Zhang JY, Rasul K, Schneider A, Ma L, Nevmyvaka Y, Song D (2025) Ts-rag: retrieval-augmented generation based time series foundation models are stronger zero-shot forecaster. arXiv preprint arXiv:2503.07649"},{"key":"335_CR336","unstructured":"Wu H, Gao Y, Shu R, Wang K, Gou R, Wu C, Liu X, He J, Cao S, Fang J et al (2025) Advanced long-term earth system forecasting by learning the small-scale nature. arXiv preprint arXiv:2505.19432"},{"key":"335_CR337","unstructured":"Yang L, Huang Z, Zhou X et al (2023) Prompt-based 3d molecular diffusion models for structure-based drug design. https:\/\/openreview.net\/forum?id=FWsGuAFn3n"},{"key":"335_CR338","doi-asserted-by":"crossref","unstructured":"Truong\u00a0Jr T, Bepler T (2024) Poet: a generative model of protein families as sequences-of-sequences. In: NeurIPS","DOI":"10.52202\/075280-3384"},{"key":"335_CR339","doi-asserted-by":"crossref","unstructured":"Frisoni G, Mizutani M, Moro G, Valgimigli L (2022) Bioreader: a retrieval-enhanced text-to-text transformer for biomedical literature. In: EMNLP","DOI":"10.18653\/v1\/2022.emnlp-main.390"},{"key":"335_CR340","doi-asserted-by":"crossref","unstructured":"Yang X, Ye M, You Q et al (2021) Writing by memorizing: hierarchical retrieval-based medical report generation. arXiv:2106.06471","DOI":"10.18653\/v1\/2021.acl-long.387"},{"key":"335_CR341","doi-asserted-by":"crossref","unstructured":"Kim J, Min M (2024) From RAG to QA-RAG: integrating generative AI for pharmaceutical regulatory compliance process. arXiv:2402.01717","DOI":"10.1145\/3672608.3707749"},{"key":"335_CR342","doi-asserted-by":"crossref","unstructured":"Ji Y, Li Z et al (2024) Rag-rlrc-laysum at biolaysumm: integrating retrieval-augmented generation and readability control for layman summarization of biomedical texts. arXiv:2405.13179","DOI":"10.18653\/v1\/2024.bionlp-1.75"},{"key":"335_CR343","unstructured":"Yang K et al (2024) Leandojo: theorem proving with retrieval-augmented language models. In: NeurIPS"},{"key":"335_CR344","unstructured":"Levonian Z, Li C, Zhu W et al (2023) Retrieval-augmented generation to improve math question-answering: trade-offs between groundedness and human preference. arXiv:2310.03184"},{"key":"335_CR345","unstructured":"Chen J, Lin H, Han X, Sun L (2023) Benchmarking large language models in retrieval-augmented generation. arxiv:2309.01431"},{"key":"335_CR346","unstructured":"ES S, James J, Anke LE, Schockaert S (2023) RAGAS: automated evaluation of retrieval augmented generation. arxiv:2309.15217"},{"key":"335_CR347","unstructured":"Saad-Falcon J, Khattab O, Potts C et al (2023) ARES: an automated evaluation framework for retrieval-augmented generation systems. arxiv:2311.09476"},{"key":"335_CR348","unstructured":"https:\/\/github.com\/truera\/trulens"},{"key":"335_CR349","unstructured":"Lyu Y, Li Z, Niu S et al (2024) CRUD-RAG: a comprehensive Chinese benchmark for retrieval-augmented generation of large language models. arxiv:2401.17043"},{"key":"335_CR350","doi-asserted-by":"crossref","unstructured":"Xiong G, Jin Q, Lu Z, Zhang A (2024) Benchmarking retrieval-augmented generation for medicine. arXiv:2402.13178","DOI":"10.18653\/v1\/2024.findings-acl.372"},{"key":"335_CR351","doi-asserted-by":"crossref","unstructured":"Petroni F, Piktus A et al (2021) Kilt: a benchmark for knowledge intensive language tasks. In: NAACL-HLT","DOI":"10.18653\/v1\/2021.naacl-main.200"},{"key":"335_CR352","first-page":"10470","volume":"37","author":"X Yang","year":"2024","unstructured":"Yang X, Sun K, Xin H, Sun Y, Bhalla N, Chen X, Choudhary S, Gui RD, Jiang ZW, Jiang Z et al (2024) Crag-comprehensive rag benchmark. Adv Neural Inf Process Syst 37:10470\u201310490","journal-title":"Adv Neural Inf Process Syst"},{"key":"335_CR353","unstructured":"Pipitone N, Alami GH (2024) Legalbench-rag: a benchmark for retrieval-augmented generation in the legal domain. arXiv preprint arXiv:2408.10343"},{"key":"335_CR354","doi-asserted-by":"crossref","unstructured":"Wang S, Tan J, Dou Z, Wen J-R (2025) Omnieval: an omnidirectional and automatic rag evaluation benchmark in financial domain. In: Proceedings of the 2025 conference on empirical methods in natural language processing, pp 5737\u20135762","DOI":"10.18653\/v1\/2025.emnlp-main.292"},{"key":"335_CR355","doi-asserted-by":"crossref","unstructured":"Barnett S, Kurniawan S, Thudumu S et al (2024) Seven failure points when engineering a retrieval augmented generation system. arXiv:2401.05856","DOI":"10.1145\/3644815.3644945"},{"key":"335_CR356","doi-asserted-by":"crossref","unstructured":"Cuconasu F, Trappolini G, Siciliano F et al (2024) The power of noise: redefining retrieval for RAG systems. arXiv:2401.14887","DOI":"10.1145\/3626772.3657834"},{"key":"335_CR357","doi-asserted-by":"crossref","unstructured":"Qiu L, Shaw P, Pasupat P et al (2022) Evaluating the impact of model scale for compositional generalization in semantic parsing. arXiv:2205.12253","DOI":"10.18653\/v1\/2022.emnlp-main.624"},{"key":"335_CR358","unstructured":"Jagerman R, Zhuang H, Qin Z et al (2023) Query expansion by prompting large language models. arxiv:2305.03653"},{"key":"335_CR359","doi-asserted-by":"crossref","unstructured":"Zhang H, Zhao P, Miao X et al (2023) Experimental analysis of large-scale learnable vector storage compression. VLDB. https:\/\/www.vldb.org\/pvldb\/vol17\/p808-zhang.pdf","DOI":"10.14778\/3636218.3636234"},{"key":"335_CR360","unstructured":"Aksitov R, Chang C, Reitter D et al (2023) Characterizing attribution and fluency tradeoffs for retrieval-augmented large language models. arXiv:2302.05578"},{"key":"335_CR361","doi-asserted-by":"crossref","unstructured":"Zhao Q, Wang R, Cen Y, Zha D, Tan S, Dong Y, Tang J (2024) Longrag: a dual-perspective retrieval-augmented generation paradigm for long-context question answering. arXiv preprint arXiv:2410.18050","DOI":"10.18653\/v1\/2024.emnlp-main.1259"},{"key":"335_CR362","unstructured":"Han C, Wang Q, Xiong W et al (2023) Lm-infinite: simple on-the-fly length generalization for large language models. arXiv:2308.16137"},{"key":"335_CR363","unstructured":"Chase H (2022) LangChain. https:\/\/github.com\/langchain-ai\/langchain"},{"key":"335_CR364","unstructured":"Jiang W, Zhang S, Han B et al (2024) Piperag: fast retrieval-augmented generation via algorithm-system co-design. arXiv:2403.05676"},{"key":"335_CR365","doi-asserted-by":"crossref","unstructured":"Hu Z, Murthy V, Pan Z, Li W, Fang X, Ding Y, Wang Y (2025) Hedrarag: co-optimizing generation and retrieval for heterogeneous rag workflows. In: Proceedings of the ACM SIGOPS 31st symposium on operating systems principles, pp 623\u2013638","DOI":"10.1145\/3731569.3764806"},{"key":"335_CR366","doi-asserted-by":"crossref","unstructured":"Ray S, Pan R, Gu Z, Du K, Feng S, Ananthanarayanan G, Netravali R, Jiang J (2025) Metis: fast quality-aware rag systems with configuration adaptation. In: Proceedings of the ACM SIGOPS 31st symposium on operating systems principles, pp 606\u2013622","DOI":"10.1145\/3731569.3764855"},{"key":"335_CR367","doi-asserted-by":"crossref","unstructured":"Yao J, Li H, Liu Y, Ray S, Cheng Y, Zhang Q, Du K, Lu S, Jiang J (2025) Cacheblend: fast large language model serving for rag with cached knowledge fusion. In: Proceedings of the twentieth European conference on computer systems, pp 94\u2013109","DOI":"10.1145\/3689031.3696098"},{"key":"335_CR368","doi-asserted-by":"crossref","unstructured":"Jin C, Zhang Z, Jiang X, Liu F, Liu S, Liu X, Jin X (2024) Ragcache: efficient knowledge caching for retrieval-augmented generation. ACM Trans Comput Syst. https:\/\/dl.acm.org\/doi\/full\/10.1145\/3768628","DOI":"10.1145\/3768628"},{"key":"335_CR369","unstructured":"Meduri K et al (2024) Efficient rag framework for large-scale knowledge bases. https:\/\/www.researchgate.net\/publication\/380265505_Efficient_RAG_Framework_for_Large-Scale_Knowledge_Bases"},{"key":"335_CR370","unstructured":"Jindal S (2024) Did Google Gemini 1.5 Really Kill RAG? https:\/\/analyticsindiamag.com\/did-google-gemini-1-5-really-kill-rag\/"},{"key":"335_CR371","unstructured":"Krishnan N (2025) Ai agents: evolution, architecture, and real-world applications. arXiv preprint arXiv:2503.12687"},{"key":"335_CR372","unstructured":"Singh A, Ehtesham A, Kumar S, Khoei TT (2025) Agentic retrieval-augmented generation: a survey on agentic rag. arXiv preprint arXiv:2501.09136"},{"key":"335_CR373","unstructured":"Xu Z, Wang M, Wang Y, Ye W, Du Y, Ma Y, Tian Y (2025) Recon: reasoning with condensation for efficient retrieval-augmented generation. arXiv preprint arXiv:2510.10448"},{"key":"335_CR374","first-page":"46534","volume":"36","author":"A Madaan","year":"2023","unstructured":"Madaan A, Tandon N, Gupta P, Hallinan S, Gao L, Wiegreffe S, Alon U, Dziri N, Prabhumoye S, Yang Y et al (2023) Self-refine: iterative refinement with self-feedback. Adv Neural Inf Process Syst 36:46534\u201346594","journal-title":"Adv Neural Inf Process Syst"},{"key":"335_CR375","first-page":"8634","volume":"36","author":"N Shinn","year":"2023","unstructured":"Shinn N, Cassano F, Gopinath A, Narasimhan K, Yao S (2023) Reflexion: language agents with verbal reinforcement learning. Adv Neural Inf Process Syst 36:8634\u20138652","journal-title":"Adv Neural Inf Process Syst"},{"key":"335_CR376","unstructured":"Wu P, Zhang M, Wan K, Zhao W, He K, Du X, Chen Z (2025) Hiprag: hierarchical process rewards for efficient agentic retrieval augmented generation. arXiv preprint arXiv:2510.07794"},{"key":"335_CR377","unstructured":"Fu Y, Peng H, Sabharwal A, Clark P, Khot T (2022) Complexity-based prompting for multi-step reasoning. arXiv preprint arXiv:2210.00720"}],"container-title":["Data Science and Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41019-025-00335-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41019-025-00335-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41019-025-00335-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T07:20:38Z","timestamp":1774250438000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s41019-025-00335-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,2]]},"references-count":377,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["335"],"URL":"https:\/\/doi.org\/10.1007\/s41019-025-00335-5","relation":{},"ISSN":["2364-1185","2364-1541"],"issn-type":[{"value":"2364-1185","type":"print"},{"value":"2364-1541","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,2]]},"assertion":[{"value":"9 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 November 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 December 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}