{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:42:17Z","timestamp":1783784537313,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":133,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731093","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:43:11Z","timestamp":1750437791000},"page":"974-989","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["RAGO: Systematic Performance Optimization for Retrieval-Augmented Generation Serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3895-7943","authenticated-orcid":false,"given":"Wenqi","family":"Jiang","sequence":"first","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8715-8964","authenticated-orcid":false,"given":"Suvinay","family":"Subramanian","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0907-583X","authenticated-orcid":false,"given":"Cat","family":"Graves","sequence":"additional","affiliation":[{"name":"Google DeepMind, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4396-6695","authenticated-orcid":false,"given":"Gustavo","family":"Alonso","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8199-7671","authenticated-orcid":false,"given":"Amir","family":"Yazdanbakhsh","sequence":"additional","affiliation":[{"name":"Google DeepMind, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1654-6684","authenticated-orcid":false,"given":"Vidushi","family":"Dadu","sequence":"additional","affiliation":[{"name":"Google, Sunnyvale, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"2021. TPU v4. https:\/\/cloud.google.com\/tpu\/docs\/v4."},{"key":"e_1_3_3_2_3_2","unstructured":"2023. TPU v5e. https:\/\/cloud.google.com\/tpu\/docs\/v5e."},{"key":"e_1_3_3_2_4_2","unstructured":"2023. TPU v5p. https:\/\/cloud.google.com\/tpu\/docs\/v5p."},{"key":"e_1_3_3_2_5_2","unstructured":"2025. Advanced RAG Techniques: Elevating Your Retrieval-Augmented Generation Systems. https:\/\/github.com\/NirDiamant\/RAG_Techniques."},{"key":"e_1_3_3_2_6_2","unstructured":"2025. ANN-Benchmarks: A Benchmarking Environment for Approximate Nearest Neighbor Algorithms Search. https:\/\/ann-benchmarks.com\/."},{"key":"e_1_3_3_2_7_2","unstructured":"2025. AWS Inferentia. https:\/\/aws.amazon.com\/ai\/machine-learning\/inferentia\/."},{"key":"e_1_3_3_2_8_2","unstructured":"2025. Faiss. https:\/\/github.com\/facebookresearch\/faiss \/."},{"key":"e_1_3_3_2_9_2","unstructured":"2025. NotebookLM: Note Taking and Research Assistant Powered by AI. https:\/\/notebooklm.google\/."},{"key":"e_1_3_3_2_10_2","unstructured":"2025. OpenAI ChatGPT. https:\/\/chat.openai.com\/."},{"key":"e_1_3_3_2_11_2","unstructured":"2025. ScaNN: Scalable Nearest Neighbors. https:\/\/github.com\/google-research\/google-research\/blob\/master\/scann."},{"key":"e_1_3_3_2_12_2","unstructured":"2025. ShareGPT: Share your ChatGPT conversations. https:\/\/sharegpt.com\/."},{"key":"e_1_3_3_2_13_2","unstructured":"2025. SIFT ANNS dataset. http:\/\/corpus-texmex.irisa.fr\/"},{"key":"e_1_3_3_2_14_2","unstructured":"Ritvik Aggarwal Ishneet Sukhvinder Singh\u00a0Ibrahim Allahverdiyev Muhammad Taha Aslihan Akalin and Kevin Zhu. 2024. ChunkRAG: Novel LLM-Chunk Filtering Method for RAG Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.19572 (2024)."},{"key":"e_1_3_3_2_15_2","volume-title":"VLDB","author":"Andr\u00e9 Fabien","year":"2016","unstructured":"Fabien Andr\u00e9, Anne-Marie Kermarrec, and Nicolas Le\u00a0Scouarnec. 2016. Cache Locality is not Enough: High-performance Nearest Neighbor Search with Product Quantization Fast Scan. In VLDB."},{"key":"e_1_3_3_2_16_2","unstructured":"Akari Asai Zeqiu Wu Yizhong Wang Avirup Sil and Hannaneh Hajishirzi. 2023. Self-RAG: Learning to Retrieve Generate and Critique through Self-Reflection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.11511 (2023)."},{"key":"e_1_3_3_2_17_2","volume-title":"CVPR","author":"Babenko Artem","year":"2016","unstructured":"Artem Babenko and Victor Lempitsky. 2016. Efficient Indexing of Billion-Scale Datasets of Deep Descriptors. In CVPR."},{"key":"e_1_3_3_2_18_2","unstructured":"Payal Bajaj Daniel Campos Nick Craswell Li Deng Jianfeng Gao Xiaodong Liu Rangan Majumder Andrew McNamara Bhaskar Mitra Tri Nguyen et\u00a0al. 2016. MS MARCO: A Human Generated MAchine Reading COmprehension Dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1611.09268 (2016)."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"D\u00e1vid Bajusz Anita R\u00e1cz and K\u00e1roly H\u00e9berger. 2015. Why is Tanimoto Index an Appropriate Choice for Fingerprint-Based Similarity Calculations? Journal of Cheminformatics (2015).","DOI":"10.1186\/s13321-015-0069-3"},{"key":"e_1_3_3_2_20_2","unstructured":"Jehyeon Bang Yujeong Choi Myeongwoo Kim Yongdeok Kim and Minsoo Rhu. 2023. vTrain: A Simulation Framework for Evaluating Cost-effective and Compute-optimal Large Language Model Training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.12391 (2023)."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Iz Beltagy Kyle Lo and Arman Cohan. 2019. SciBERT: A Pretrained Language Model for Scientific Text. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1903.10676 (2019).","DOI":"10.18653\/v1\/D19-1371"},{"key":"e_1_3_3_2_22_2","unstructured":"Maciej Besta Ales Kubicek Roman Niggli Robert Gerstenberger Lucas Weitzendorf Mingyuan Chi Patrick Iff Joanna Gajda Piotr Nyczyk J\u00fcrgen M\u00fcller et\u00a0al. 2024. Multi-Head RAG: Solving Multi-Aspect Problems with LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.05085 (2024)."},{"key":"e_1_3_3_2_23_2","volume-title":"ICML","author":"Borgeaud Sebastian","year":"2022","unstructured":"Sebastian Borgeaud, Arthur Mensch, Jordan Hoffmann, Trevor Cai, Eliza Rutherford, Katie Millican, George\u00a0Bm Van Den\u00a0Driessche, Jean-Baptiste Lespiau, Bogdan Damoc, Aidan Clark, et\u00a0al. 2022. Improving Language Models by Retrieving From Trillions of Tokens. In ICML."},{"key":"e_1_3_3_2_24_2","volume-title":"NeurIPS","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et\u00a0al. 2020. Language Models are Few-shot Learners. In NeurIPS."},{"key":"e_1_3_3_2_25_2","unstructured":"Chi-Min Chan Chunpu Xu Ruibin Yuan Hongyin Luo Wei Xue Yike Guo and Jie Fu. 2024. RQ-RAG: Learning to Refine Queries for Retrieval Augmented Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.00610 (2024)."},{"key":"e_1_3_3_2_26_2","unstructured":"Qi Chen Bing Zhao Haidong Wang Mingqin Li Chuanjie Liu Zengzhong Li Mao Yang and Jingdong Wang. 2021. SPANN: Highly-efficient Billion-scale Approximate Nearest Neighbor Search. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.08566 (2021)."},{"key":"e_1_3_3_2_27_2","unstructured":"Aakanksha Chowdhery Sharan Narang Jacob Devlin Maarten Bosma Gaurav Mishra Adam Roberts Paul Barham Hyung\u00a0Won Chung Charles Sutton Sebastian Gehrmann et\u00a0al. 2022. PaLM: Scaling Language Modeling with Pathways. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.02311 (2022)."},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_3_2_29_2","unstructured":"Databricks. 2024. RAG (Retrieval Augmented Generation) on Databricks."},{"key":"e_1_3_3_2_30_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et\u00a0al. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_2_31_2","unstructured":"Darren Edge Ha Trinh Newman Cheng Joshua Bradley Alex Chao Apurva Mody Steven Truitt and Jonathan Larson. 2024. From Local to Global: A Graph RAG Approach to Query-Focused Summarization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.16130 (2024)."},{"key":"e_1_3_3_2_32_2","unstructured":"Angela Fan Yacine Jernite Ethan Perez David Grangier Jason Weston and Michael Auli. 2019. ELI5: Long form Question Answering. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1907.09190 (2019)."},{"key":"e_1_3_3_2_33_2","unstructured":"Cong Fu Chao Xiang Changxu Wang and Deng Cai. 2017. Fast Approximate Nearest Neighbor Search with the Navigating Spreading-out Graph. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1707.00143 (2017)."},{"key":"e_1_3_3_2_34_2","unstructured":"Jianyang Gao and Cheng Long. 2023. High-dimensional Approximate Nearest Neighbor Search: With Reliable and Efficient Distance Comparison Operations. Proceedings of the ACM on Management of Data (2023)."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2008.4563100"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Michael Glass Gaetano Rossiello Md\u00a0Faisal\u00a0Mahbub Chowdhury Ankita\u00a0Rajaram Naik Pengshan Cai and Alfio Gliozzo. 2022. Re2G: Retrieve Rerank Generate. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2207.06300 (2022).","DOI":"10.18653\/v1\/2022.naacl-main.194"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Fabian Groh Lukas Ruppert Patrick Wieschollek and Hendrik\u00a0PA Lensch. 2022. GGNN: Graph-Based GPU Nearest Neighbor Search. IEEE Transactions on Big Data (2022).","DOI":"10.1109\/TBDATA.2022.3161156"},{"key":"e_1_3_3_2_38_2","volume-title":"ICML","author":"Guo Ruiqi","year":"2020","unstructured":"Ruiqi Guo, Philip Sun, Erik Lindgren, Quan Geng, David Simcha, Felix Chern, and Sanjiv Kumar. 2020. Accelerating Large-Scale Inference with Anisotropic Vector Quantization. In ICML."},{"key":"e_1_3_3_2_39_2","volume-title":"ICML","author":"Guu Kelvin","year":"2020","unstructured":"Kelvin Guu, Kenton Lee, Zora Tung, Panupong Pasupat, and Mingwei Chang. 2020. Retrieval Augmented Language Model Pre-training. In ICML."},{"key":"e_1_3_3_2_40_2","unstructured":"Kelvin Guu Kenton Lee Zora Tung Panupong Pasupat and Ming-Wei Chang. 2020. REALM: Retrieval-Augmented Language Model Pre-Training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2002.08909 (2020)."},{"key":"e_1_3_3_2_41_2","volume-title":"ISCA","author":"Huang Qijing","year":"2024","unstructured":"Qijing Huang, Po-An Tsai, Joel\u00a0S Emer, and Angshuman Parashar. 2024. Mind the Gap: Attainable Data Movement and Operational Intensity Bounds for Tensor Algorithms. In ISCA."},{"key":"e_1_3_3_2_42_2","volume-title":"NeurIPS","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc\u00a0V Le, Yonghui Wu, et\u00a0al. 2019. GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism. In NeurIPS."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Gautier Izacard and Edouard Grave. 2020. Leveraging Passage Retrieval with Generative Models for Open Domain Question Answering. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2007.01282 (2020).","DOI":"10.18653\/v1\/2021.eacl-main.74"},{"key":"e_1_3_3_2_44_2","volume-title":"ATC","author":"Jang Junhyeok","year":"2023","unstructured":"Junhyeok Jang, Hanjin Choi, Hanyeoreum Bae, Seungjun Lee, Miryeong Kwon, and Myoungsoo Jung. 2023. CXL-ANNS: Software-Hardware Collaborative Memory Disaggregation and Computation for Billion-Scale Approximate Nearest Neighbor Search. In ATC."},{"key":"e_1_3_3_2_45_2","unstructured":"Suhas Jayaram\u00a0Subramanya Fnu Devvrit Harsha\u00a0Vardhan Simhadri Ravishankar Krishnawamy and Rohan Kadekodi. 2019. DiskANN: Fast Accurate Billion-point Nearest Neighbor Search on a Single Node. NeurIPS."},{"key":"e_1_3_3_2_46_2","unstructured":"Herve Jegou Matthijs Douze and Cordelia Schmid. 2010. Product Quantization for Nearest Neighbor Search. IEEE Transactions on Pattern Analysis and Machine Intelligence (2010)."},{"key":"e_1_3_3_2_47_2","unstructured":"Ziwei Ji Nayeon Lee Rita Frieske Tiezheng Yu Dan Su Yan Xu Etsuko Ishii Ye\u00a0Jin Bang Andrea Madotto and Pascale Fung. 2023. Survey of Hallucination in Natural Language Generation. Comput. Surveys (2023)."},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467139"},{"key":"e_1_3_3_2_49_2","unstructured":"Wenqi Jiang Hang Hu Torsten Hoefler and Gustavo Alonso. 2024. Accelerating Graph-based Vector Search via Delayed-Synchronization Traversal. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.12385 (2024)."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607045"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"crossref","unstructured":"Wenqi Jiang Marco Zeller Roger Waleffe Torsten Hoefler and Gustavo Alonso. 2025. Chameleon: A Heterogeneous and Disaggregated Accelerator System for Retrieval-Augmented Language Models. VLDB (2025).","DOI":"10.14778\/3696435.3696439"},{"key":"e_1_3_3_2_52_2","volume-title":"KDD","author":"Jiang Wenqi","year":"2025","unstructured":"Wenqi Jiang, Shuai Zhang, Boran Han, Jie Wang, Bernie Wang, and Tim Kraska. 2025. PipeRAG: Fast Retrieval-Augmented Generation via Algorithm-System Co-design. In KDD."},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"crossref","unstructured":"Zhengbao Jiang Frank\u00a0F Xu Luyu Gao Zhiqing Sun Qian Liu Jane Dwivedi-Yu Yiming Yang Jamie Callan and Graham Neubig. 2023. Active Retrieval Augmented Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.06983 (2023).","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"key":"e_1_3_3_2_54_2","unstructured":"Chao Jin Zili Zhang Xuanlin Jiang Fangyue Liu Xin Liu Xuanzhe Liu and Xin Jin. 2024. RAGCache: Efficient Knowledge Caching for Retrieval-Augmented Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.12457 (2024)."},{"key":"e_1_3_3_2_55_2","unstructured":"Jeff Johnson Matthijs Douze and Herv\u00e9 J\u00e9gou. 2019. Billion-Scale Similarity Search with GPUs. IEEE Transactions on Big Data (2019)."},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"crossref","unstructured":"Mandar Joshi Eunsol Choi Daniel\u00a0S Weld and Luke Zettlemoyer. 2017. TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1705.03551 (2017).","DOI":"10.18653\/v1\/P17-1147"},{"key":"e_1_3_3_2_57_2","volume-title":"ISCA","author":"Jouppi Norman\u00a0P","year":"2017","unstructured":"Norman\u00a0P Jouppi, Cliff Young, Nishant Patil, David Patterson, Gaurav Agrawal, Raminder Bajwa, Sarah Bates, Suresh Bhatia, Nan Boden, Al Borchers, et\u00a0al. 2017. In-Datacenter Performance Analysis of a Tensor Processing Unit. In ISCA."},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"crossref","unstructured":"Enkelejda Kasneci Kathrin Se\u00dfler Stefan K\u00fcchemann Maria Bannert Daryna Dementieva Frank Fischer Urs Gasser Georg Groh Stephan G\u00fcnnemann Eyke H\u00fcllermeier et\u00a0al. 2023. ChatGPT for Good? On Opportunities and Challenges of Large Language Models for Education. Learning and Individual Differences (2023).","DOI":"10.35542\/osf.io\/5er8f"},{"key":"e_1_3_3_2_59_2","unstructured":"Liu Ke Xuan Zhang Benjamin Lee G\u00a0Edward Suh and Hsien-Hsin\u00a0S Lee. 2022. DisaggRec: Architecting Disaggregated Systems for Large-Scale Personalized Recommendation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2212.00939 (2022)."},{"key":"e_1_3_3_2_60_2","unstructured":"Urvashi Khandelwal Angela Fan Dan Jurafsky Luke Zettlemoyer and Mike Lewis. 2020. Nearest Neighbor Machine Translation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.00710 (2020)."},{"key":"e_1_3_3_2_61_2","unstructured":"Urvashi Khandelwal Omer Levy Dan Jurafsky Luke Zettlemoyer and Mike Lewis. 2019. Generalization through Memorization: Nearest Neighbor Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1911.00172 (2019)."},{"key":"e_1_3_3_2_62_2","unstructured":"Mojtaba Komeili Kurt Shuster and Jason Weston. 2021. Internet-Augmented Dialogue Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.07566 (2021)."},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_64_2","volume-title":"NeurIPS","author":"Lazaridou Angeliki","year":"2021","unstructured":"Angeliki Lazaridou, Adhi Kuncoro, Elena Gribovskaya, Devang Agrawal, Adam Liska, Tayfun Terzi, Mai Gimenez, Cyprien de Masson\u00a0d\u2019Autume, Tomas Kocisky, Sebastian Ruder, et\u00a0al. 2021. Mind the Gap: Assessing Temporal Generalization in Neural Language Models. In NeurIPS."},{"key":"e_1_3_3_2_65_2","unstructured":"Jinhyuk Lee Anthony Chen Zhuyun Dai Dheeru Dua Devendra\u00a0Singh Sachan Michael Boratko Yi Luan S\u00e9bastien\u00a0MR Arnold Vincent Perot Siddharth Dalmia et\u00a0al. 2024. Can Long-Context Language Models Subsume Retrieval RAG SQL and More? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.13121 (2024)."},{"key":"e_1_3_3_2_66_2","unstructured":"Jinhyuk Lee Zhuyun Dai Xiaoqi Ren Blair Chen Daniel Cer Jeremy\u00a0R Cole Kai Hui Michael Boratko Rajvi Kapadia Wen Ding et\u00a0al. 2024. Gecko: Versatile Text Embeddings Distilled From Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.20327 (2024)."},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"crossref","unstructured":"Jungi Lee Wonbeom Lee and Jaewoong Sim. 2024. Tender: Accelerating Large Language Models via Tensor Decomposition and Runtime Requantization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.12930 (2024).","DOI":"10.1109\/ISCA59077.2024.00080"},{"key":"e_1_3_3_2_68_2","unstructured":"Jinhyuk Lee Wonjin Yoon Sungdong Kim Donghyeon Kim Sunkyu Kim Chan\u00a0Ho So and Jaewoo Kang. 2020. BioBERT: A Pre-trained Biomedical Language Representation Model for Biomedical Text Mining. Bioinformatics (2020)."},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00021"},{"key":"e_1_3_3_2_70_2","unstructured":"Alexandria Leto Cecilia Aguerrebere Ishwar Bhati Ted Willke Mariano Tepper and Vy\u00a0Ai Vo. 2024. Toward Optimal Search and Retrieval for RAG. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.07396 (2024)."},{"key":"e_1_3_3_2_71_2","volume-title":"NeurIPS","author":"Lewis Mike","year":"2020","unstructured":"Mike Lewis, Marjan Ghazvininejad, Gargi Ghosh, Armen Aghajanyan, Sida Wang, and Luke Zettlemoyer. 2020. Pre-training via Paraphrasing. In NeurIPS."},{"key":"e_1_3_3_2_72_2","volume-title":"NeurIPS","author":"Lewis Patrick","year":"2020","unstructured":"Patrick Lewis, Ethan Perez, Aleksandra Piktus, Fabio Petroni, Vladimir Karpukhin, Naman Goyal, Heinrich K\u00fcttler, Mike Lewis, Wen-tau Yih, Tim Rockt\u00e4schel, et\u00a0al. 2020. Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks. In NeurIPS."},{"key":"e_1_3_3_2_73_2","unstructured":"Jinhao Li Jiaming Xu Shan Huang Yonghua Chen Wen Li Jun Liu Yaoxiu Lian Jiayi Pan Li Ding Hao Zhou et\u00a0al. 2024. Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.04466 (2024)."},{"key":"e_1_3_3_2_74_2","unstructured":"Yujia Li David Choi Junyoung Chung Nate Kushman Julian Schrittwieser R\u00e9mi Leblond Tom Eccles James Keeling Felix Gimeno Agustin Dal\u00a0Lago et\u00a0al. 2022. Competition-Level Code Generation with AlphaCode. Science (2022)."},{"key":"e_1_3_3_2_75_2","unstructured":"Zihao Li. 2023. The Dark Side of ChatGPT: Legal and Ethical Challenges From Stochastic Parrots and Hallucination. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.14347 (2023)."},{"key":"e_1_3_3_2_76_2","unstructured":"Zhuowan Li Cheng Li Mingyang Zhang Qiaozhu Mei and Michael Bendersky. 2024. Retrieval Augmented Generation or Long-context LLMs? A Comprehensive Study and Hybrid Approach. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.16833 (2024)."},{"key":"e_1_3_3_2_77_2","volume-title":"NeurIPS","author":"Liu Jiawei","year":"2024","unstructured":"Jiawei Liu, Chunqiu\u00a0Steven Xia, Yuyao Wang, and Lingming Zhang. 2024. Is your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation. In NeurIPS."},{"key":"e_1_3_3_2_78_2","unstructured":"Zihan Liu Wentao Ni Jingwen Leng Yu Feng Cong Guo Quan Chen Chao Li Minyi Guo and Yuhao Zhu. 2023. JUNO: Optimizing High-Dimensional Approximate Nearest Neighbour Search with Sparsity-Aware Algorithm and Ray-Tracing Core Mapping. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.01712 (2023)."},{"key":"e_1_3_3_2_79_2","unstructured":"Kejing Lu Mineichi Kudo Chuan Xiao and Yoshiharu Ishikawa. 2021. HVS: Hierarchical Graph Structure based on Voronoi Diagrams for Solving Approximate Nearest Neighbor Search. VLDB (2021)."},{"key":"e_1_3_3_2_80_2","unstructured":"Xinbei Ma Yeyun Gong Pengcheng He Hai Zhao and Nan Duan. 2023. Query Rewriting for Retrieval-Augmented Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.14283 (2023)."},{"key":"e_1_3_3_2_81_2","doi-asserted-by":"crossref","unstructured":"Divya Mahajan Joon\u00a0Kyung Kim Jacob Sacks Adel Ardalan Arun Kumar and Hadi Esmaeilzadeh. 2018. In-RDBMS Hardware Acceleration of Advanced Analytics. VLDB (2018).","DOI":"10.14778\/3236187.3236188"},{"key":"e_1_3_3_2_82_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358320"},{"key":"e_1_3_3_2_83_2","doi-asserted-by":"crossref","unstructured":"Yury Malkov Alexander Ponomarenko Andrey Logvinov and Vladimir Krylov. 2014. Approximate Nearest Neighbor Algorithm based on Navigable Small World Graphs. Information Systems (2014).","DOI":"10.1016\/j.is.2013.10.006"},{"key":"e_1_3_3_2_84_2","unstructured":"Yu\u00a0A Malkov and Dmitry\u00a0A Yashunin. 2018. Efficient and Robust Approximate Nearest Neighbor Search using Hierarchical Navigable Small World Graphs. IEEE Transactions on Pattern Analysis and Machine Intelligence (2018)."},{"key":"e_1_3_3_2_85_2","unstructured":"Meta. 2024. Build AI Knowledge Assistants over your Enterprise Data."},{"key":"e_1_3_3_2_86_2","doi-asserted-by":"crossref","unstructured":"Jason Mohoney Anil Pacaci Shihabur\u00a0Rahman Chowdhury Ali Mousavi Ihab\u00a0F Ilyas Umar\u00a0Farooq Minhas Jeffrey Pound and Theodoros Rekatsinas. 2023. High-Throughput Vector Similarity Search in Knowledge Graphs. Proceedings of the ACM on Management of Data (2023).","DOI":"10.1145\/3589777"},{"key":"e_1_3_3_2_87_2","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_3_2_88_2","unstructured":"James\u00a0Jie Pan Jianguo Wang and Guoliang Li. 2023. Survey of Vector Database Management Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.14021 (2023)."},{"key":"e_1_3_3_2_89_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2019.00042"},{"key":"e_1_3_3_2_90_2","unstructured":"Pratyush Patel Esha Choukse Chaojie Zhang \u00cd\u00f1igo Goiri Aashaka Shah Saeed Maleki and Ricardo Bianchini. 2023. Splitwise: Efficient Generative LLM Inference Using Phase Splitting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.18677 (2023)."},{"key":"e_1_3_3_2_91_2","unstructured":"Gabriel Poesia Oleksandr Polozov Vu Le Ashish Tiwari Gustavo Soares Christopher Meek and Sumit Gulwani. 2022. Synchromesh: Reliable Code Generation From Pre-trained Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2201.11227 (2022)."},{"key":"e_1_3_3_2_92_2","volume-title":"ISCA","author":"Qin Yubin","year":"2024","unstructured":"Yubin Qin, Yang Wang, Zhiren Zhao, Xiaolong Yang, Yang Zhou, Shaojun Wei, Yang Hu, and Shouyi Yin. 2024. MECLA: Memory-Compute-Efficient LLM Accelerator with Scaling Sub-matrix Partition. In ISCA."},{"key":"e_1_3_3_2_93_2","unstructured":"Jack\u00a0W Rae Sebastian Borgeaud Trevor Cai Katie Millican Jordan Hoffmann Francis Song John Aslanides Sarah Henderson Roman Ring Susannah Young et\u00a0al. 2021. Scaling Language Models: Methods Analysis & Insights From Training Gopher. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2112.11446 (2021)."},{"key":"e_1_3_3_2_94_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_3_2_95_2","doi-asserted-by":"crossref","unstructured":"Pranav Rajpurkar Robin Jia and Percy Liang. 2018. Know What You Don\u2019t Know: Unanswerable Questions for SQuAD. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1806.03822 (2018).","DOI":"10.18653\/v1\/P18-2124"},{"key":"e_1_3_3_2_96_2","doi-asserted-by":"crossref","unstructured":"Ori Ram Yoav Levine Itay Dalmedigos Dor Muhlgay Amnon Shashua Kevin Leyton-Brown and Yoav Shoham. 2023. In-Context Retrieval-Augmented Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.00083 (2023).","DOI":"10.1162\/tacl_a_00605"},{"key":"e_1_3_3_2_97_2","doi-asserted-by":"crossref","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1908.10084 (2019).","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_3_2_98_2","unstructured":"Michael Ruminer. 2024. Google\u2019s NotebookLM and RAG."},{"key":"e_1_3_3_2_99_2","doi-asserted-by":"crossref","unstructured":"Jie Shao Zi Huang Heng\u00a0Tao Shen Xiaofang Zhou Ee-Peng Lim and Yijun Li. 2008. Batch Nearest Neighbor Search for Video Retrieval. IEEE Transactions on Multimedia (2008).","DOI":"10.1109\/TMM.2008.917339"},{"key":"e_1_3_3_2_100_2","unstructured":"Rulin Shao Jacqueline He Akari Asai Weijia Shi Tim Dettmers Sewon Min Luke Zettlemoyer and Pang\u00a0Wei Koh. 2024. Scaling Retrieval-Based Language Models with a Trillion-Token Datastore. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.12854 (2024)."},{"key":"e_1_3_3_2_101_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358302"},{"key":"e_1_3_3_2_102_2","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1909.08053 (2019)."},{"key":"e_1_3_3_2_103_2","unstructured":"Harsha\u00a0Vardhan Simhadri George Williams Martin Aum\u00fcller Matthijs Douze Artem Babenko Dmitry Baranchuk Qi Chen Lucas Hosseini Ravishankar Krishnaswamy Gopal Srinivasa et\u00a0al. 2022. Results of the NeurIPS\u201921 Challenge on Billion-Scale Approximate Nearest Neighbor Search. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.03763 (2022)."},{"key":"e_1_3_3_2_104_2","unstructured":"Shaden Smith Mostofa Patwary Brandon Norick Patrick LeGresley Samyam Rajbhandari Jared Casper Zhun Liu Shrimai Prabhumoye George Zerveas Vijay Korthikanti et\u00a0al. 2022. Using DeepSpeed and Megatron to Train Megatron-Turing NLG 530B A Large-Scale Generative Language Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2201.11990 (2022)."},{"key":"e_1_3_3_2_105_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15286-3_16"},{"key":"e_1_3_3_2_106_2","unstructured":"Philip Sun Ruiqi Guo and Sanjiv Kumar. 2023. Automating Nearest Neighbor Search Configuration with Constrained Optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.01702 (2023)."},{"key":"e_1_3_3_2_107_2","volume-title":"NeurIPS","author":"Sun Philip","year":"2024","unstructured":"Philip Sun, David Simcha, Dave Dopson, Ruiqi Guo, and Sanjiv Kumar. 2024. SOAR: Improved Indexing for Approximate Nearest Neighbor Search. In NeurIPS."},{"key":"e_1_3_3_2_108_2","unstructured":"Ross Taylor Marcin Kardas Guillem Cucurull Thomas Scialom Anthony Hartshorn Elvis Saravia Andrew Poulton Viktor Kerkez and Robert Stojnic. 2022. Galactica: A Large Language Model for Science. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.09085 (2022)."},{"key":"e_1_3_3_2_109_2","unstructured":"Gemini Team Petko Georgiev Ving\u00a0Ian Lei Ryan Burnell Libin Bai Anmol Gulati Garrett Tanzer Damien Vincent Zhufeng Pan Shibo Wang et\u00a0al. 2024. Gemini 1.5: Unlocking Multimodal Understanding Across Millions of Tokens of Context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_2_110_2","doi-asserted-by":"crossref","unstructured":"Arun\u00a0James Thirunavukarasu Darren Shu\u00a0Jeng Ting Kabilan Elangovan Laura Gutierrez Ting\u00a0Fang Tan and Daniel Shu\u00a0Wei Ting. 2023. Large language Models in Medicine. Nature medicine (2023).","DOI":"10.1038\/s41591-023-02448-8"},{"key":"e_1_3_3_2_111_2","unstructured":"Harsh Trivedi Niranjan Balasubramanian Tushar Khot and Ashish Sabharwal. 2022. Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2212.10509 (2022)."},{"key":"e_1_3_3_2_112_2","volume-title":"NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention Is All You Need. In NeurIPS."},{"key":"e_1_3_3_2_113_2","unstructured":"Boxin Wang Wei Ping Lawrence McAfee Peng Xu Bo Li Mohammad Shoeybi and Bryan Catanzaro. 2023. InstructRetro: Instruction Tuning post Retrieval-Augmented Pretraining. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.07713 (2023)."},{"key":"e_1_3_3_2_114_2","unstructured":"Shuting Wang Xin Xu Mang Wang Weipeng Chen Yutao Zhu and Zhicheng Dou. 2024. RichRAG: Crafting Rich Responses for Multi-faceted Queries in Retrieval-Augmented Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.12566 (2024)."},{"key":"e_1_3_3_2_115_2","doi-asserted-by":"crossref","unstructured":"Chuangxian Wei Bin Wu Sheng Wang Renjie Lou Chaoqun Zhan Feifei Li and Yuanzhe Cai. 2020. AnalyticDB-V: a hybrid analytical engine towards query fusion for structured and unstructured data. Proceedings of the VLDB Endowment 13 12 (2020) 3152\u20133165.","DOI":"10.14778\/3415478.3415541"},{"key":"e_1_3_3_2_116_2","doi-asserted-by":"crossref","unstructured":"Jonathan Woodbridge Bobak Mortazavi Alex\u00a0AT Bui and Majid Sarrafzadeh. 2016. Improving Biomedical Signal Search Results in Big Data Case-Based Reasoning Environments. Pervasive and mobile computing (2016).","DOI":"10.1016\/j.pmcj.2015.09.006"},{"key":"e_1_3_3_2_117_2","unstructured":"Frank\u00a0F Xu Uri Alon and Graham Neubig. 2023. Why do Nearest Neighbor Language Models Work? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.02828 (2023)."},{"key":"e_1_3_3_2_118_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613166"},{"key":"e_1_3_3_2_119_2","unstructured":"Xiao Yang Kai Sun Hao Xin Yushi Sun Nikita Bhalla Xiangsen Chen Sajal Choudhary Rongze\u00a0Daniel Gui Ziran\u00a0Will Jiang Ziyu Jiang et\u00a0al. 2024. CRAG\u2013Comprehensive RAG Benchmark. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.04744 (2024)."},{"key":"e_1_3_3_2_120_2","unstructured":"Jiayi Yao Hanchen Li Yuhan Liu Siddhant Ray Yihua Cheng Qizheng Zhang Kuntai Du Shan Lu and Junchen Jiang. 2024. CacheBlend: Fast Large Language Model Serving with Cached Knowledge Fusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.16444 (2024)."},{"key":"e_1_3_3_2_121_2","volume-title":"OSDI","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu, Joo\u00a0Seong Jeong, Geon-Woo Kim, Soojeong Kim, and Byung-Gon Chun. 2022. Orca: A Distributed Serving System for Transformer-Based Generative Models. In OSDI."},{"key":"e_1_3_3_2_122_2","unstructured":"Yue Yu Wei Ping Zihan Liu Boxin Wang Jiaxuan You Chao Zhang Mohammad Shoeybi and Bryan Catanzaro. 2024. RankRAG: Unifying Context Ranking with Retrieval-Augmented Generation in LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.02485 (2024)."},{"key":"e_1_3_3_2_123_2","unstructured":"Zhenrui Yue Honglei Zhuang Aijun Bai Kai Hui Rolf Jagerman Hansi Zeng Zhen Qin Dong Wang Xuanhui Wang and Michael Bendersky. 2024. Inference Scaling for Long-Context Retrieval Augmented Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.04343 (2024)."},{"key":"e_1_3_3_2_124_2","unstructured":"Sungmin Yun Kwanhee Kyung Juhwan Cho Jaewan Choi Jongmin Kim Byeongho Kim Sukhan Lee Kyomin Sohn and Jung\u00a0Ho Ahn. 2024. Duplex: A Device for Large Language Models with Mixture of Experts Grouped Query Attention and Continuous Batching. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.01141 (2024)."},{"key":"e_1_3_3_2_125_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614292"},{"key":"e_1_3_3_2_126_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507767"},{"key":"e_1_3_3_2_127_2","volume-title":"ISCA","author":"Zhang Hengrui","year":"2024","unstructured":"Hengrui Zhang, August Ning, Rohan\u00a0Baskar Prabhakar, and David Wentzlaff. 2024. LLMCompass: Enabling Efficient Hardware Design for Large Language Model Inference. In ISCA."},{"key":"e_1_3_3_2_128_2","volume-title":"OSDI","author":"Zhang Qianxi","year":"2023","unstructured":"Qianxi Zhang, Shuotao Xu, Qi Chen, Guoxin Sui, Jiadong Xie, Zhizhen Cai, Yaoqi Chen, Yinxuan He, Yuqing Yang, Fan Yang, et\u00a0al. 2023. VBASE: Unifying Online Vector Similarity Search and Relational Queries via Relaxed Monotonicity. In OSDI."},{"key":"e_1_3_3_2_129_2","unstructured":"Zhihao Zhang Alan Zhu Lijie Yang Yihua Xu Lanting Li Phitchaya\u00a0Mangpo Phothilimthana and Zhihao Jia. 2024. Accelerating Retrieval-Augmented Language Model Serving with Speculation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.14021 (2024)."},{"key":"e_1_3_3_2_130_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE48307.2020.00094"},{"key":"e_1_3_3_2_131_2","doi-asserted-by":"crossref","unstructured":"Xi Zhao Yao Tian Kai Huang Bolong Zheng and Xiaofang Zhou. 2023. Towards Efficient Index Construction and Approximate Nearest Neighbor Search in High-Dimensional Spaces. VLDB (2023).","DOI":"10.14778\/3594512.3594527"},{"key":"e_1_3_3_2_132_2","doi-asserted-by":"crossref","unstructured":"Youpeng Zhao Di Wu and Jun Wang. 2024. ALISA: Accelerating Large Language Model Inference via Sparsity-Aware KV Caching. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.17312 (2024).","DOI":"10.1109\/ISCA59077.2024.00077"},{"key":"e_1_3_3_2_133_2","volume-title":"OSDI","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In OSDI."},{"key":"e_1_3_3_2_134_2","unstructured":"Chaoji Zuo and Dong Deng. 2023. ARKGraph: All-Range Approximate K-Nearest-Neighbor Graph. VLDB (2023)."}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731093","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:05:08Z","timestamp":1750503908000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731093"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":133,"alternative-id":["10.1145\/3695053.3731093","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731093","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}