{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:10:48Z","timestamp":1785953448129,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731032","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:43:11Z","timestamp":1750437791000},"page":"450-466","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["In-Storage Acceleration of Retrieval Augmented Generation as a Service"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2887-9761","authenticated-orcid":false,"given":"Rohan","family":"Mahapatra","sequence":"first","affiliation":[{"name":"University of California San Diego, La Jolla, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1636-5430","authenticated-orcid":false,"given":"Harsha","family":"Santhanam","sequence":"additional","affiliation":[{"name":"University of California San Diego, La Jolla, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4028-6448","authenticated-orcid":false,"given":"Christopher","family":"Priebe","sequence":"additional","affiliation":[{"name":"University of California San Diego, La Jolla, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0328-9610","authenticated-orcid":false,"given":"Hanyang","family":"Xu","sequence":"additional","affiliation":[{"name":"University of California San Diego, La Jolla, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8548-1039","authenticated-orcid":false,"given":"Hadi","family":"Esmaeilzadeh","sequence":"additional","affiliation":[{"name":"University of California San Diego, La Jolla, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Ahmet\u00a0Yasin Aytar Kemal Kilic and Kamer Kaya. 2024. A Retrieval-Augmented Generation Framework for Academic Literature Navigation in Data Science. arXiv (2024)."},{"key":"e_1_3_3_3_3_2","volume-title":"ICML","author":"Cai Tianle","year":"2024","unstructured":"Tianle Cai, Yuhong Li, Zhengyang Geng, Hongwu Peng, Jason\u00a0D. Lee, Deming Chen, and Tri Dao. 2024. MEDUSA: Simple LLM inference acceleration framework with multiple decoding heads. In ICML."},{"key":"e_1_3_3_3_4_2","volume-title":"NeurIPS","author":"Chen Qi","year":"2021","unstructured":"Qi Chen, Bing Zhao, Haidong Wang, Mingqin Li, Chuanjie Liu, Zengzhong Li, Mao Yang, and Jingdong Wang. 2021. Spann: Highly-efficient billion-scale approximate nearest neighborhood search. In NeurIPS."},{"key":"e_1_3_3_3_5_2","volume-title":"NeurIPS","author":"Chen Qi","year":"2021","unstructured":"Qi Chen, Bing Zhao, Haidong Wang, Mingqin Li, Chuanjie Liu, Zengzhong Li, Mao Yang, and Jingdong Wang. 2021. Spann: Highly-efficient billion-scale approximate nearest neighborhood search. In NeurIPS."},{"key":"e_1_3_3_3_6_2","unstructured":"Rongxin Cheng Yifan Peng Xingda Wei Hongrui Xie Rong Chen Sijie Shen and Haibo Chen. 2024. Characterizing the Dilemma of Performance and Index Size in Billion-Scale Vector Search and Breaking It with Second-Tier Memory. arXiv (2024)."},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"crossref","unstructured":"Ha Cho Tae Jun Y.H. Kim Hee Kang Imjin Ahn Hansle Gwon Yunha Kim Hyeram Seo Heejung Choi Minkyoung Kim JiYe Han Gaeun Kee Seohyun Park and Soyoung Ko. 2024. Task-Specific Transformer-Based Language Models in Health Care: Scoping Review. JMIR Medical Informatics 12 (2024).","DOI":"10.2196\/49724"},{"key":"e_1_3_3_3_8_2","volume-title":"ICLR","author":"Dao Tri","year":"2024","unstructured":"Tri Dao. 2024. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. In ICLR."},{"key":"e_1_3_3_3_9_2","volume-title":"NeurIPS","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Daniel\u00a0Y. Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. FLASHATTENTION: fast and memory-efficient exact attention with IO-awareness. In NeurIPS."},{"key":"e_1_3_3_3_10_2","unstructured":"Matthew Davis Mark Dustin and Rahul Agarwal. 2023. Accelerate generative AI workloads on Amazon Aurora with optimized reads and pgvector. https:\/\/aws.amazon.com\/blogs\/database\/accelerate-generative-ai-workloads-on-amazon-aurora-with-optimized-reads-and-pgvector\/."},{"key":"e_1_3_3_3_11_2","volume-title":"NAACL","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL."},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/2463676.2465295"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"publisher","DOI":"10.14778\/3503585.3503594"},{"key":"e_1_3_3_3_14_2","unstructured":"Matthijs Douze Alexandr Guzhva Chengqi Deng Jeff Johnson Gergely Szilvasy Pierre-Emmanuel Mazar\u00e9 Maria Lomeli Lucas Hosseini and Herv\u00e9 J\u00e9gou. 2024. The faiss library. arXiv (2024)."},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441578"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Thibault Formal Carlos Lassance Benjamin Piwowarski and St\u00e9phane Clinchant. 2021. SPLADE v2: Sparse Lexical and Expansion Model for Information Retrieval. arXiv (2021).","DOI":"10.1145\/3404835.3463098"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463098"},{"key":"e_1_3_3_3_18_2","unstructured":"Chris Fregly. 2024. Amazon Bedrock Retrieval-Augmented Generation (RAG) Workshop. https:\/\/github.com\/aws-samples\/amazon-bedrock-rag-workshop."},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.14778\/3303753.3303754"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.379"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640365"},{"key":"e_1_3_3_3_22_2","unstructured":"Google. 2025. Google AI for Developers. https:\/\/ai.google.dev\/gemini-api\/docs\/long-context."},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.14778\/3554821.3554843"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISQED54688.2022.9806270"},{"key":"e_1_3_3_3_25_2","unstructured":"Jennifer Hsia Afreen Shaikh Zhiruo Wang and Graham Neubig. 2024. RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems. arXiv (2024)."},{"key":"e_1_3_3_3_26_2","volume-title":"NeurIPS","author":"Huang Haiyang","year":"2024","unstructured":"Haiyang Huang, Newsha Ardalani, Anna Sun, Liu Ke, Shruti Bhosale, Hsien-Hsin\u00a0S. Lee, Carole-Jean Wu, and Benjamin Lee. 2024. Toward Efficient Inference for Mixture of Experts. In NeurIPS."},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"crossref","unstructured":"Lei Huang Weijiang Yu Weitao Ma Weihong Zhong Zhangyin Feng Haotian Wang Qianglong Chen Weihua Peng Xiaocheng Feng Bing Qin and Ting Liu. 2025. A Survey on Hallucination in Large Language Models: Principles Taxonomy Challenges and Open Questions. ACM Trans. Inf. Syst. 43 2 Article 42 (2025).","DOI":"10.1145\/3703155"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"Suyeon Hur Seongmin Na Dongup Kwon Joonsung Kim Andrew Boutros Eriko Nurvitadhi and Jangwoo Kim. 2023. A Fast and Flexible FPGA-based Accelerator for Natural Language Processing Neural Networks. ACM Trans. Archit. Code Optim. 20 1 (2023).","DOI":"10.1145\/3564606"},{"key":"e_1_3_3_3_29_2","volume-title":"ISCA","author":"Hwang Ranggi","year":"2024","unstructured":"Ranggi Hwang, Jianyu Wei, Shijie Cao, Changho Hwang, Xiaohu Tang, Ting Cao, and Mao Yang. 2024. Pre-gated MoE: An Algorithm-System Co-Design for Fast and Scalable Mixture-of-Expert Inference. In ISCA."},{"key":"e_1_3_3_3_30_2","volume-title":"ISCA","author":"Jang Hanhwi","year":"2019","unstructured":"Hanhwi Jang, Joonsung Kim, Jae-Eon Jo, Jaewon Lee, and Jangwoo Kim. 2019. Mnnfast: A fast and scalable system architecture for memory-augmented neural networks. In ISCA."},{"key":"e_1_3_3_3_31_2","volume-title":"NeurIPS","author":"Jayaram\u00a0Subramanya Suhas","year":"2019","unstructured":"Suhas Jayaram\u00a0Subramanya, Fnu Devvrit, Harsha\u00a0Vardhan Simhadri, Ravishankar Krishnawamy, and Rohan Kadekodi. 2019. Diskann: Fast accurate billion-point nearest neighbor search on a single node. In NeurIPS."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.91"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"publisher","DOI":"10.14778\/3696435.3696439"},{"key":"e_1_3_3_3_34_2","volume-title":"KDD","author":"Jiang Wenqi","year":"2025","unstructured":"Wenqi Jiang, Shuai Zhang, Boran Han, Jie Wang, Bernie Wang, and Tim Kraska. 2025. PipeRAG: Fast retrieval-augmented generation via algorithm-system co-design. In KDD."},{"key":"e_1_3_3_3_35_2","volume-title":"ISCA","author":"Jouppi Norman\u00a0P.","year":"2021","unstructured":"Norman\u00a0P. Jouppi, Doe Hyun\u00a0Yoon, Matthew Ashcraft, Mark Gottscho, Thomas\u00a0B. Jablin, George Kurian, James Laudon, Sheng Li, Peter Ma, Xiaoyu Ma, Thomas Norrie, Nishant Patil, Sushma Prasad, Cliff Young, Zongwei Zhou, and David Patterson. 2021. Ten Lessons From Three Generations Shaped Google\u2019s TPUv4i : Industrial Product. In ISCA."},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/3431920.3439477"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"crossref","unstructured":"Ji-Hoon Kim Yeo-Reum Park Jaeyoung Do Soo-Young Ji and Joo-Young Kim. 2022. Accelerating large-scale graph-based nearest neighbor search on a computational storage platform. IEEE Trans. Comput. 72 1 (2022).","DOI":"10.1109\/TC.2022.3155956"},{"key":"e_1_3_3_3_39_2","unstructured":"Inc. Kioxia\u00a0America. 2024. How to Accelerate Vector Databases Without High DRAM Costs: Use Disk-Based Vector Indexes with Fast PCIe 5.0 SSDs from Kioxia. https:\/\/blog-us.kioxia.com\/post\/2024\/11\/12\/how-to-accelerate-vector-databases-without-high-dram-costs-use-disk-based-vector-indexes-with-fast-pcie-5-0-ssds-from-kioxia."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"crossref","unstructured":"Anastasia Krithara Anastasios Nentidis Konstantinos Bougiatiotis and Georgios Paliouras. 2023. BioASQ-QA: A manually curated corpus for Biomedical Question Answering. Scientific Data 10 1 (2023).","DOI":"10.1038\/s41597-023-02068-4"},{"key":"e_1_3_3_3_41_2","volume-title":"ISCA","author":"Kwon Youngeun","year":"2022","unstructured":"Youngeun Kwon and Minsoo Rhu. 2022. Training Personalized Recommendation Systems from (GPU) Scratch: Look Forward Not Backwards. In ISCA."},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W16-1609"},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"crossref","unstructured":"Joo\u00a0Hwan Lee Hui Zhang Veronica Lagrange Praveen Krishnamoorthy Xiaodong Zhao and Yang\u00a0Seok Ki. 2020. SmartSSD: FPGA Accelerated Near-Storage Data Analytics on SSD. IEEE Comput. Archit. Lett. 19 2 (2020).","DOI":"10.1109\/LCA.2020.3009347"},{"key":"e_1_3_3_3_44_2","volume-title":"OSDI","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. InfiniGen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In OSDI."},{"key":"e_1_3_3_3_45_2","volume-title":"NeurIPS","author":"Lewis Patrick","year":"2020","unstructured":"Patrick Lewis, Ethan Perez, Aleksandra Piktus, Fabio Petroni, Vladimir Karpukhin, Naman Goyal, Heinrich K\u00fcttler, Mike Lewis, Wen-tau Yih, Tim Rockt\u00e4schel, Sebastian Riedel, and Douwe Kiela. 2020. Retrieval-augmented generation for knowledge-intensive NLP tasks. In NeurIPS."},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3370748.3406567"},{"key":"e_1_3_3_3_47_2","volume-title":"COLING","author":"Li Siran","year":"2025","unstructured":"Siran Li, Linus Stenzel, Carsten Eickhoff, and Seyed\u00a0Ali Bahrainian. 2025. Enhancing Retrieval-Augmented Generation: A Study of Best Practices. In COLING."},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"crossref","unstructured":"Wen Li Ying Zhang Yifang Sun Wei Wang Mingjie Li Wenjie Zhang and Xuemin Lin. 2019. Approximate nearest neighbor search on high dimensional data\u2014experiments analyses and improvement. IEEE Trans. on Knowl. and Data Eng. 32 8 (2019).","DOI":"10.1109\/TKDE.2019.2909204"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463238"},{"key":"e_1_3_3_3_50_2","volume-title":"MLSys","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Jiaming Tang, Haotian Tang, Shang Yang, Wei-Ming Chen, Wei-Chen Wang, Guangxuan Xiao, Xingyu Dang, Chuang Gan, and Song Han. 2024. AWQ: Activation-aware Weight Quantization for On-Device LLM Compression and Acceleration. In MLSys."},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640360"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480125"},{"key":"e_1_3_3_3_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/SOCC49529.2020.9524802"},{"key":"e_1_3_3_3_54_2","volume-title":"NeurIPS","author":"Lu Zepu","year":"2024","unstructured":"Zepu Lu, Jin Chen, Defu Lian, Zaixi Zhang, Yong Ge, and Enhong Chen. 2024. Knowledge distillation for high dimensional search index. In NeurIPS."},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615112"},{"key":"e_1_3_3_3_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640413"},{"key":"e_1_3_3_3_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3352460.3358320"},{"key":"e_1_3_3_3_58_2","doi-asserted-by":"crossref","unstructured":"Yu\u00a0A. Malkov and D.\u00a0A. Yashunin. 2020. Efficient and Robust Approximate Nearest Neighbor Search Using Hierarchical Navigable Small World Graphs. IEEE Trans. Pattern Anal. Mach. Intell. 42 4 (2020).","DOI":"10.1109\/TPAMI.2018.2889473"},{"key":"e_1_3_3_3_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507702"},{"key":"e_1_3_3_3_60_2","doi-asserted-by":"crossref","unstructured":"Jing Miao Charat Thongprayoon Supawadee Suppadungsuk Oscar\u00a0A. Garcia\u00a0Valencia and Wisit Cheungpasitporn. 2024. Integrating Retrieval-Augmented Generation with Large Language Models in Nephrology: Advancing Practical Applications. Medicina 60 3 (2024).","DOI":"10.3390\/medicina60030445"},{"key":"e_1_3_3_3_61_2","unstructured":"Microsoft. 2024. RAG and Generative AI - Azure AI Search. https:\/\/learn.microsoft.com\/en-us\/azure\/search\/retrieval-augmented-generation-overview."},{"key":"e_1_3_3_3_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2007.33"},{"key":"e_1_3_3_3_63_2","unstructured":"National Library of Medicine. 2023. MEDLINE\/PubMed Baseline Repository (MBR). https:\/\/lhncbc.nlm.nih.gov\/ii\/information\/MBR.html."},{"key":"e_1_3_3_3_64_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.146"},{"key":"e_1_3_3_3_65_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.669"},{"key":"e_1_3_3_3_66_2","unstructured":"Nvidia. 2024. NVIDIA System Management Interface (nvidia-smi) Documentation. https:\/\/docs.nvidia.com\/deploy\/nvidia-smi\/index.html."},{"key":"e_1_3_3_3_67_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.15"},{"key":"e_1_3_3_3_68_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640422"},{"key":"e_1_3_3_3_69_2","unstructured":"Xiao Peng and Liang Chen. 2024. Athena: Retrieval-augmented Legal Judgment Prediction with Large Language Models. arXiv (2024)."},{"key":"e_1_3_3_3_70_2","doi-asserted-by":"crossref","unstructured":"Shahzad Qaiser and Ramsha Ali. 2018. Text mining: use of TF-IDF to examine the relevance of words to documents. International Journal of Computer Applications 181 1 (2018).","DOI":"10.5120\/ijca2018917395"},{"key":"e_1_3_3_3_71_2","volume-title":"ISCA","author":"Qin Yubin","year":"2023","unstructured":"Yubin Qin, Yang Wang, Dazheng Deng, Zhiren Zhao, Xiaolong Yang, Leibo Liu, Shaojun Wei, Yang Hu, and Shouyi Yin. 2023. FACT: FFN-Attention Co-optimized Transformer Architecture with Eager Correlation Prediction. In ISCA."},{"key":"e_1_3_3_3_72_2","volume-title":"ISCA","author":"Qin Yubin","year":"2024","unstructured":"Yubin Qin, Yang Wang, Zhiren Zhao, Xiaolong Yang, Yang Zhou, Shaojun Wei, Yang Hu, and Shouyi Yin. 2024. MECLA: Memory-Compute-Efficient LLM Accelerator with Scaling Sub-matrix Partition. In ISCA."},{"key":"e_1_3_3_3_73_2","doi-asserted-by":"publisher","DOI":"10.1145\/3669940.3707264"},{"key":"e_1_3_3_3_74_2","doi-asserted-by":"crossref","unstructured":"Ori Ram Yoav Levine Itay Dalmedigos Dor Muhlgay Amnon Shashua Kevin Leyton-Brown and Yoav Shoham. 2023. In-context retrieval-augmented language models. Transactions of the Association for Computational Linguistics 11 (2023).","DOI":"10.1162\/tacl_a_00605"},{"key":"e_1_3_3_3_75_2","volume-title":"ISCA","author":"Rashidi Saeed","year":"2022","unstructured":"Saeed Rashidi, William Won, Sudarshan Srinivasan, Srinivas Sridharan, and Tushar Krishna. 2022. Themis: a network bandwidth-aware collective scheduling policy for distributed training of DL models. In ISCA."},{"key":"e_1_3_3_3_76_2","volume-title":"NeurIPS","author":"Ren Jie","year":"2020","unstructured":"Jie Ren, Minjia Zhang, and Dong Li. 2020. HM-ANN: efficient billion-point nearest neighbor search on heterogeneous memory. In NeurIPS."},{"key":"e_1_3_3_3_77_2","doi-asserted-by":"crossref","unstructured":"Stephen Robertson and Hugo Zaragoza. 2009. The Probabilistic Relevance Framework: BM25 and Beyond. Found. Trends Inf. Retr. 3 4 (2009).","DOI":"10.1561\/1500000019"},{"key":"e_1_3_3_3_78_2","doi-asserted-by":"crossref","unstructured":"Efraim Rotem Alon Naveh Avinash Ananthakrishnan Eliezer Weissmann and Doron Rajwan. 2012. Power-Management Architecture of the Intel Microarchitecture Code-Named Sandy Bridge. IEEE Micro 32 2 (2012).","DOI":"10.1109\/MM.2012.12"},{"key":"e_1_3_3_3_79_2","unstructured":"Amazon SageMaker. 2024. Question Answering with Cohere and LangChain using JumpStart. https:\/\/sagemaker-examples.readthedocs.io\/en\/latest\/introduction_to_amazon_algorithms\/jumpstart-foundation-models\/question_answering_retrieval_augmented_generation\/question_answering_Cohere+langchain_jumpstart.html."},{"key":"e_1_3_3_3_80_2","unstructured":"Amazon SageMaker. 2024. Question Answering with Pinecone and LLaMA-2 using JumpStart. https:\/\/sagemaker-examples.readthedocs.io\/en\/latest\/introduction_to_amazon_algorithms\/jumpstart-foundation-models\/question_answering_retrieval_augmented_generation\/question_answering_pinecone_llama-2_jumpstart.html."},{"key":"e_1_3_3_3_81_2","unstructured":"Amazon\u00a0Web Services. 2024. Amazon Bedrock Knowledge Bases. https:\/\/aws.amazon.com\/bedrock\/knowledge-bases\/."},{"key":"e_1_3_3_3_82_2","unstructured":"Amazon\u00a0Web Services. 2024. What is OpenSearch? https:\/\/aws.amazon.com\/what-is\/opensearch\/."},{"key":"e_1_3_3_3_83_2","unstructured":"Michael Shen Muhammad Umar Kiwan Maeng G.\u00a0Edward Suh and Udit Gupta. 2024. Towards Understanding Systems Trade-offs in Retrieval-Augmented Generation Model Inference. arXiv (2024)."},{"key":"e_1_3_3_3_84_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.320"},{"key":"e_1_3_3_3_85_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian\u00a0Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit\u00a0Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric\u00a0Michael Smith Ranjan Subramanian Xiaoqing\u00a0Ellen Tan Binh Tang Ross Taylor Adina Williams Jian\u00a0Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. https:\/\/arxiv.org\/abs\/2307.09288. arXiv (2023)."},{"key":"e_1_3_3_3_86_2","volume-title":"NeurIPS","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In NeurIPS."},{"key":"e_1_3_3_3_87_2","doi-asserted-by":"crossref","unstructured":"Hanyin Wang Chufan Gao Christopher Dantona Bryan Hull and Jimeng Sun. 2024. DRG-LLaMA: tuning LLaMA model to predict diagnosis-related group for hospitalized patients. npj Digital Medicine 7 1 (2024).","DOI":"10.1038\/s41746-023-00989-3"},{"key":"e_1_3_3_3_88_2","doi-asserted-by":"crossref","unstructured":"Mengzhao Wang Weizhi Xu Xiaomeng Yi Songlin Wu Zhangyang Peng Xiangyu Ke Yunjun Gao Xiaoliang Xu Rentong Guo and Charles Xie. 2024. Starling: An I\/O-Efficient Disk-Resident Graph Index Framework for High-Dimensional Vector Similarity Search on Data Segment. Proc. ACM Manag. Data 2 1 (2024).","DOI":"10.1145\/3639269"},{"key":"e_1_3_3_3_89_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00083"},{"key":"e_1_3_3_3_90_2","volume-title":"ISCA","author":"Wang Yitu","year":"2024","unstructured":"Yitu Wang, Shiyu Li, Qilin Zheng, Linghao Song, Zongwang Li, Andrew Chang, Hai\u00a0\"Helen\" Li, and Yiran Chen. 2024. NDSEARCH: Accelerating Graph-Traversal-Based Approximate Nearest Neighbor Search through Near Data Processing. In ISCA."},{"key":"e_1_3_3_3_91_2","doi-asserted-by":"publisher","DOI":"10.1145\/2757667.2757684"},{"key":"e_1_3_3_3_92_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613166"},{"key":"e_1_3_3_3_93_2","volume-title":"MICRO","author":"Yu Zhongkai","year":"2024","unstructured":"Zhongkai Yu, Shengwen Liang, Tianyun Ma, Yunke Cai, Ziyuan Nan, Di Huang, Xinkai Song, Yifan Hao, Jie Zhang, Tian Zhi, Yongwei Zhao, Zidong Du, Xing Hu, Qi Guo, and Tianshi Chen. 2024. FlashLLM: A Chiplet-Based In-Flash Computing Architecture to Enable On-Device Inference of 70B LLM. In MICRO."},{"key":"e_1_3_3_3_94_2","doi-asserted-by":"publisher","DOI":"10.1145\/3604237.3626866"},{"key":"e_1_3_3_3_95_2","volume-title":"WWW","author":"Zhao Xuejiao","year":"2025","unstructured":"Xuejiao Zhao, Siyan Liu, Su-Yin Yang, and Chunyan Miao. 2025. MedRAG: Enhancing Retrieval-augmented Generation with Knowledge Graph-Elicited Reasoning for Healthcare Copilot. In WWW."}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731032","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T10:59:52Z","timestamp":1750503592000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731032"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":94,"alternative-id":["10.1145\/3695053.3731032","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731032","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}